{"id":"W4413114846","doi":"10.1101/2025.08.09.25333360","title":"Diagnostic Codes in AI prediction models and Label Leakage of Same-admission Clinical Outcomes","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Leakage (economics); Computer science; Reliability engineering; Artificial intelligence; Medicine; Engineering; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07239413,0.0008533026,0.001093405,0.008909611,0.0004647054,0.00393883,0.001419023,0.001063047,0.0008679066],"category_scores_gemma":[0.2720765,0.0004842736,0.003738317,0.007129316,0.001104411,0.002462803,0.002105774,0.001576742,0.0001801494],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001195728,"about_ca_system_score_gemma":0.002684222,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002246118,"about_ca_topic_score_gemma":0.002379827,"domain_scores_codex":[0.9266934,0.04945927,0.01132085,0.004205617,0.007672387,0.0006484575],"domain_scores_gemma":[0.5666388,0.3714109,0.04080193,0.007174894,0.01333409,0.0006393084],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001743196,0.0001880095,0.8167489,0.006571046,0.009466747,0.0003426803,0.0004939855,0.01698128,0.0005415439,0.002359614,0.002387968,0.142175],"study_design_scores_gemma":[0.001097175,0.002697607,0.4973759,0.01812029,0.03455152,0.002329067,0.002168458,0.3752348,0.009081184,0.04232087,0.01466633,0.0003568234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7735416,0.08658467,0.1175671,0.007420657,0.0008484206,0.0009077951,0.007646739,0.0004756071,0.005007303],"genre_scores_gemma":[0.981119,0.002698534,0.01391297,0.0004241731,0.0002059851,0.0001896576,0.001323132,0.00001701726,0.000109604],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9276059,"threshold_uncertainty_score":0.3828613,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04275382199184965,"score_gpt":0.3868823508937452,"score_spread":0.3441285289018955,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}