{"id":"W7117294252","doi":"10.1001/jamanetworkopen.2025.50454","title":"Diagnostic Codes in AI Prediction Models and Label Leakage of Same-Admission Clinical Outcomes","year":2025,"lang":"en","type":"article","venue":"JAMA Network Open","topic":"Sepsis Diagnosis and Treatment","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"U.S. National Library of Medicine","keywords":"Predictive modelling; Trustworthiness; Multiple Models; Clinical Practice; Diagnosis code; Artificial neural network; Interpretability","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07879021,0.0009499993,0.001142212,0.005387671,0.000747449,0.004668928,0.00157786,0.001624674,0.0011217],"category_scores_gemma":[0.3353713,0.000515586,0.002468403,0.006076796,0.002175661,0.004334919,0.00258554,0.002211419,0.0002621091],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001387018,"about_ca_system_score_gemma":0.002211875,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001915647,"about_ca_topic_score_gemma":0.001938361,"domain_scores_codex":[0.9241069,0.05033697,0.010573,0.004498556,0.009822446,0.0006620376],"domain_scores_gemma":[0.5477827,0.359475,0.05793455,0.01701095,0.01705427,0.0007425315],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0017325,0.0002166427,0.7548673,0.005221544,0.004185156,0.0003236093,0.001334048,0.009295266,0.0008108604,0.005133432,0.001963162,0.2149165],"study_design_scores_gemma":[0.0006661873,0.003737453,0.6882721,0.02072001,0.013817,0.00327007,0.003285857,0.1567149,0.01659326,0.07015418,0.02233153,0.0004373995],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7629209,0.06477029,0.1490468,0.01036931,0.00119875,0.0005809953,0.003863515,0.0003669394,0.006882568],"genre_scores_gemma":[0.9764397,0.003085279,0.01778643,0.0008575288,0.0003781585,0.0002206076,0.0009456429,0.00002943996,0.0002571381],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9212098,"threshold_uncertainty_score":0.4166874,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1324866201201463,"score_gpt":0.4361251982803647,"score_spread":0.3036385781602184,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}