{"id":"W4406259803","doi":"10.1109/bibm62325.2024.10822432","title":"Gene function prediction using an AnnoTree-based genomic language model","year":2024,"lang":"en","type":"article","venue":"","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Wilfrid Laurier University; University of Waterloo","funders":"","keywords":"Computer science; Function (biology); Language model; Artificial intelligence; Natural language processing; Computational biology; Biology; Genetics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004061898,0.0006165413,0.0003933119,0.0008628265,0.0002585642,0.0005369664,0.0006011165,0.0005161982,0.001966626],"category_scores_gemma":[0.001090864,0.0001914216,0.0008943318,0.0005315667,0.0002331056,0.0005971242,0.0005639935,0.0006757044,0.0009733624],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005208607,"about_ca_system_score_gemma":0.0006131174,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00561192,"about_ca_topic_score_gemma":0.008605831,"domain_scores_codex":[0.9997463,0.00007125625,0.00001618374,0.00009063879,0.00004417411,0.00003137269],"domain_scores_gemma":[0.9996532,0.0001841804,0.00002467241,0.0000337837,0.00008406177,0.00002012878],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001152859,0.0005120503,0.01888069,0.0004591631,0.00024996,0.001140832,0.0003678689,0.5784861,0.1035263,0.02401251,0.01931436,0.2518972],"study_design_scores_gemma":[0.00001512827,0.0000535504,0.0006540161,0.000006646768,0.00001397453,0.00008279704,0.00002729971,0.9888816,0.004187138,0.004069025,0.001997652,0.0000110052],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3029556,0.0004851194,0.6689026,0.0005713397,0.00009836403,0.0001048397,0.01224793,0.01105217,0.00358216],"genre_scores_gemma":[0.7385954,0.0003182182,0.227312,0.0003184,0.00004659119,0.0002100121,0.02917897,0.0006454245,0.003374989],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00561192,"threshold_uncertainty_score":0.01115853,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01251341949451856,"score_gpt":0.2651069880560296,"score_spread":0.2525935685615111,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}