{"id":"W4417474898","doi":"10.64898/2025.12.16.25342438","title":"A medically grounded LLM agent–based tool to detect patient safety events in medical records","year":2025,"lang":"en","type":"article","venue":"medRxiv","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Hallucinating; Patient safety; Probabilistic logic; Medical record; Adverse effect; Clinical decision making; Key (lock); MEDLINE","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002012151,0.0002104918,0.0003210609,0.0003837596,0.0001266977,0.00003355644,0.001580048,0.0001846733,0.0002148029],"category_scores_gemma":[0.003644884,0.0001916023,0.00008806379,0.001428454,0.0000347707,0.0001081356,0.0005824043,0.0006479073,0.00009915247],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003030709,"about_ca_system_score_gemma":0.0007543638,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005477868,"about_ca_topic_score_gemma":0.001091065,"domain_scores_codex":[0.9960638,0.0007105476,0.0007439612,0.000710046,0.001213122,0.0005585084],"domain_scores_gemma":[0.9980048,0.0005440859,0.00009741136,0.0008942054,0.00009483431,0.0003646412],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001041269,0.0001820542,0.214354,0.0001426761,0.00001810645,0.000244156,0.000631169,0.0002727559,0.00003573314,0.003065109,0.0008043039,0.7801458],"study_design_scores_gemma":[0.001609931,0.000586133,0.8092788,0.0007897487,0.000006306693,0.00001315461,0.00001624597,0.1312351,0.0001774256,0.003020289,0.05281688,0.0004499343],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6039804,0.0000579302,0.3609261,0.03083482,0.001703682,0.0005923852,0.000001988181,0.0002551574,0.001647581],"genre_scores_gemma":[0.9640929,0.000009683306,0.02356699,0.01196194,0.00005444168,0.0001177812,0.000003976103,0.0000141686,0.0001781993],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7796959,"threshold_uncertainty_score":0.7813314,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01267384909878684,"score_gpt":0.2998256056496219,"score_spread":0.287151756550835,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}