{"id":"W4224284596","doi":"10.1101/2022.04.15.22273900","title":"EHR Foundation Models Improve Robustness in the Presence of Temporal Distribution Shift","year":2022,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Clinical Evaluative Sciences; SickKids Foundation; Hospital for Sick Children","funders":"","keywords":"Logistic regression; Robustness (evolution); Receiver operating characteristic; Computer science; Artificial intelligence; Transformer; Machine learning; Statistics; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007559767,0.001056178,0.0008467598,0.0009113033,0.0003303149,0.001308538,0.001079115,0.0008290226,0.001251799],"category_scores_gemma":[0.02747049,0.0004428257,0.001014429,0.0005323358,0.0005257505,0.002019998,0.00174675,0.002192311,0.0005983285],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001046336,"about_ca_system_score_gemma":0.00131635,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.010405,"about_ca_topic_score_gemma":0.009334255,"domain_scores_codex":[0.9981552,0.0008213548,0.0001291088,0.0005339278,0.0001876416,0.0001727383],"domain_scores_gemma":[0.9871755,0.008757946,0.001137142,0.001428343,0.001185121,0.0003158298],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001053994,0.0005626988,0.1248066,0.0001620375,0.0005916815,0.0002245344,0.0003160756,0.6507002,0.003011511,0.001542999,0.005297835,0.2117298],"study_design_scores_gemma":[0.00002791355,0.000201803,0.006513346,0.00002866237,0.00004626302,0.0000501549,0.00004146607,0.989895,0.0009131048,0.001884841,0.0003794678,0.00001792491],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8443962,0.000990689,0.1456026,0.001967868,0.0001434183,0.00013084,0.001222054,0.002895942,0.002650376],"genre_scores_gemma":[0.9879473,0.0001199101,0.0101418,0.0001923899,0.00003658466,0.00003344023,0.001012948,0.00004717484,0.0004684679],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.010405,"threshold_uncertainty_score":0.03998029,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03788948645225321,"score_gpt":0.3111427006191986,"score_spread":0.2732532141669454,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}