{"id":"W4408387458","doi":"10.1038/s42256-025-01007-9","title":"Transformers and genome language models","year":2025,"lang":"en","type":"article","venue":"Nature Machine Intelligence","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":75,"is_retracted":false,"has_abstract":false,"ca_institutions":"Lunenfeld-Tanenbaum Research Institute; Vector Institute; Public Health Ontario; University of Toronto; University Health Network","funders":"","keywords":"Computer science; Transformer; Computational biology; Genome; Biology; Gene; Genetics; Engineering; Electrical engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001613011,0.000717517,0.0008777982,0.002033453,0.001228028,0.004590945,0.001374595,0.001795471,0.01131605],"category_scores_gemma":[0.01010667,0.0009107973,0.001663666,0.002461175,0.00333359,0.01535181,0.002115626,0.003609328,0.002434367],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001634821,"about_ca_system_score_gemma":0.0008890938,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003075415,"about_ca_topic_score_gemma":0.002132917,"domain_scores_codex":[0.9985448,0.0006302022,0.0001012755,0.0003376269,0.0002438304,0.00014231],"domain_scores_gemma":[0.9939254,0.004528464,0.0001990473,0.000811169,0.0003893154,0.0001465861],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001235817,0.000004581068,0.00007852321,0.00001686174,0.000005009128,0.00002034678,0.0001009871,0.001762783,0.000092613,0.9906785,0.0013009,0.005926452],"study_design_scores_gemma":[0.000003080934,0.000001336688,0.00001451335,0.000004250699,0.000003485839,0.00001714809,0.00001710882,0.006899978,0.0001049933,0.9905288,0.002401957,0.000003421907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02848455,0.002034454,0.9146457,0.008867627,0.0002982263,0.00005769917,0.001625786,0.002305752,0.04168022],"genre_scores_gemma":[0.8256345,0.00252008,0.1384576,0.001838017,0.000709217,0.0001879148,0.003213428,0.001249013,0.02619018],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01131605,"threshold_uncertainty_score":0.03785598,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.004233397503920612,"score_gpt":0.258521215793566,"score_spread":0.2542878182896454,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}