{"id":"W4410060872","doi":"10.1038/s41746-025-01589-z","title":"Expert of Experts Verification and Alignment (EVAL) Framework for Large Language Models Safety in Gastroenterology","year":2025,"lang":"en","type":"article","venue":"npj Digital Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"National Institute of Diabetes and Digestive and Kidney Diseases; National Institutes of Health","keywords":"Computer science; Medical physics; Medicine; Natural language processing; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002164978,0.0000927108,0.0002783294,0.0001470312,0.00002793505,0.000004571303,0.00004683848,0.00009671654,0.00002354974],"category_scores_gemma":[0.0005195109,0.00007591834,0.00002755745,0.000125284,0.00007707917,0.00009427281,0.00001812532,0.00007607159,7.936796e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008138163,"about_ca_system_score_gemma":0.00005392367,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000130598,"about_ca_topic_score_gemma":0.00003164939,"domain_scores_codex":[0.9990118,0.00001730883,0.0004442204,0.000206196,0.0001192968,0.0002012234],"domain_scores_gemma":[0.9993338,0.0002398063,0.00006912564,0.0002014787,0.00008383577,0.00007197914],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.008322283,0.002044887,0.1412239,0.001631994,0.0001788155,0.00002130253,0.1007363,0.00004607398,0.01252214,0.1580328,0.008056688,0.5671828],"study_design_scores_gemma":[0.009308242,0.01505967,0.08362339,0.01851605,0.0003953187,0.0001194028,0.4035195,0.0366634,0.06942568,0.3179254,0.04433639,0.001107519],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8761424,0.002079638,0.1029171,0.01540398,0.0004558854,0.0008568006,0.00001904905,0.00002538141,0.002099786],"genre_scores_gemma":[0.9965671,0.0002791432,0.0008464677,0.001818252,0.0001473955,0.00008488478,0.00007883632,0.000008305363,0.0001696238],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5660753,"threshold_uncertainty_score":0.309586,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0731186697190006,"score_gpt":0.4205977627554838,"score_spread":0.3474790930364832,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}