{"id":"W4224284596","doi":"10.1101/2022.04.15.22273900","title":"EHR Foundation Models Improve Robustness in the Presence of Temporal Distribution Shift","year":2022,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Clinical Evaluative Sciences; SickKids Foundation; Hospital for Sick Children","funders":"","keywords":"Logistic regression; Robustness (evolution); Receiver operating characteristic; Computer science; Artificial intelligence; Transformer; Machine learning; Statistics; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00253094,0.0002075829,0.0002845562,0.0001112796,0.0001402029,0.0001306479,0.002920561,0.000144536,0.00002410992],"category_scores_gemma":[0.0003114434,0.0001715224,0.00009528027,0.000495488,0.00006320779,0.0002933688,0.002045545,0.001387359,0.000002094421],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001626865,"about_ca_system_score_gemma":0.0003425383,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002279676,"about_ca_topic_score_gemma":0.0001778309,"domain_scores_codex":[0.9965938,0.001103299,0.0005505542,0.0006780979,0.0007866696,0.0002876294],"domain_scores_gemma":[0.9974893,0.0003378397,0.0004972334,0.001525413,0.0001062459,0.00004393127],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001930304,0.0001350703,0.1564209,0.0007149515,0.00001206788,0.00002489223,0.00522767,0.7555574,0.00001253135,0.07338093,0.0000632896,0.008430957],"study_design_scores_gemma":[0.0001077374,0.00005235354,0.07131626,0.00006474357,0.000004510421,0.000002505087,0.00005580034,0.9083053,0.00001578351,0.01957992,0.0003313599,0.0001636948],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4411347,0.00009019745,0.5527914,0.003962644,0.001034652,0.0006869818,0.00003524654,0.00007747966,0.0001867199],"genre_scores_gemma":[0.9958008,0.00001483027,0.003512658,0.00005772634,0.00007650835,0.0002600941,0.0002329599,0.00001183022,0.00003259613],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5546661,"threshold_uncertainty_score":0.699448,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03788948645225321,"score_gpt":0.3111427006191986,"score_spread":0.2732532141669454,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}