{"id":"W4415116989","doi":"10.3390/bioengineering12101098","title":"Disease-Specific Prediction of Missense Variant Pathogenicity with DNA Language Models and Graph Neural Networks","year":2025,"lang":"en","type":"article","venue":"Bioengineering","topic":"RNA and protein synthesis mechanisms","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; McGill University Health Centre","funders":"Canada First Research Excellence Fund","keywords":"ENCODE; Pathogenicity; Missense mutation; Convolutional neural network; Artificial neural network; Human genome; Classifier (UML); Genomics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006591172,0.0008930878,0.0003901472,0.00131102,0.000179597,0.0004870103,0.0005879942,0.0006937178,0.00070743],"category_scores_gemma":[0.002839132,0.0002790356,0.0006398765,0.0005613921,0.0003151417,0.000755896,0.000405051,0.0007425811,0.0002440149],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008874498,"about_ca_system_score_gemma":0.0005140924,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01182878,"about_ca_topic_score_gemma":0.01721705,"domain_scores_codex":[0.9997432,0.0001000732,0.00001563702,0.00007170958,0.00003752527,0.00003186299],"domain_scores_gemma":[0.9985716,0.001012378,0.000150445,0.00006869281,0.000152919,0.00004380061],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001734984,0.0001157095,0.007155,0.00004130733,0.00009460627,0.0001757604,0.00002915311,0.9339217,0.004385053,0.001947525,0.0009982485,0.05096242],"study_design_scores_gemma":[0.000002948494,0.000008314179,0.0002329276,0.0000013002,0.000004219054,0.000008383224,0.000002269881,0.9977307,0.0003668492,0.001602726,0.00003708565,0.000002219398],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5117282,0.0007139529,0.4809072,0.001185033,0.00008456731,0.0000833719,0.001148041,0.002450753,0.001698769],"genre_scores_gemma":[0.9428346,0.0001695591,0.05449346,0.0001416849,0.00003201372,0.00004088257,0.0009891309,0.00005887438,0.001239753],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01182878,"threshold_uncertainty_score":0.02351987,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008387351122493973,"score_gpt":0.1890641949045245,"score_spread":0.1806768437820306,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}