{"id":"W4405625946","doi":"10.1093/bioinformatics/btag415","title":"Fitness translocation: improving variant effect prediction with biologically-grounded data augmentation","year":2024,"lang":"en","type":"preprint","venue":"Bioinformatics","topic":"RNA and protein synthesis mechanisms","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Bottleneck; Fitness function; Computer science; Fitness landscape; Sequence (biology); Function (biology); Machine learning; Selection (genetic algorithm); Computational biology; Artificial intelligence; Biology; Genetics; Medicine; Genetic algorithm; Population","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0007258175,0.0003508698,0.0002756098,0.0000725279,0.0001018707,0.0002358012,0.0006132376,0.0005459122,0.00001948507],"category_scores_gemma":[0.00006769133,0.0002525753,0.00007980318,0.00008784762,0.00006258694,0.00001637373,0.0008162822,0.0002711382,0.00003161389],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003715442,"about_ca_system_score_gemma":0.0002345596,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002920592,"about_ca_topic_score_gemma":0.00001474512,"domain_scores_codex":[0.9984203,0.00009294652,0.0004988112,0.000502886,0.0002517584,0.0002332773],"domain_scores_gemma":[0.9983913,0.00002659198,0.0002627553,0.001168666,0.00007547099,0.00007518339],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001283256,0.0001943268,0.00009996998,0.008687719,0.001509411,0.00003784657,0.0009689135,0.002037315,0.2913992,0.0009306229,0.002893205,0.6899582],"study_design_scores_gemma":[0.004156947,0.006378142,0.0005063579,0.002973552,0.001885868,0.0005132994,0.0008143974,0.2272853,0.7284545,0.004132207,0.0196892,0.003210186],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1488582,0.00236754,0.8385136,0.0003947527,0.002156115,0.003084363,0.001854805,0.0002724756,0.002498167],"genre_scores_gemma":[0.9036783,0.0006219892,0.0749552,0.0001882008,0.0009192338,0.0004250656,0.01878686,0.00008499423,0.0003401471],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7635584,"threshold_uncertainty_score":0.9999927,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02251479689590924,"score_gpt":0.2593764694314939,"score_spread":0.2368616725355847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}