{"id":"W2965485927","doi":"10.48550/arxiv.1908.00690","title":"Feature Robustness in Non-stationary Health Records: Caveats to Deployable Model Performance in Common Clinical Machine Learning Tasks","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":55,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Robustness (evolution); Computer science; Health records; Artificial intelligence; Machine learning; Feature (linguistics); Health care","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0291933,0.001301428,0.001125146,0.0009132896,0.0009802791,0.002586575,0.003280902,0.001741775,0.002254394],"category_scores_gemma":[0.1818027,0.0009532433,0.001246282,0.00142059,0.002192942,0.004320838,0.002931458,0.004655587,0.001851094],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001238335,"about_ca_system_score_gemma":0.001470743,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01491649,"about_ca_topic_score_gemma":0.007281722,"domain_scores_codex":[0.9880905,0.005393841,0.001222009,0.002260178,0.002432685,0.0006007219],"domain_scores_gemma":[0.9152829,0.05285121,0.002733756,0.02297612,0.00553521,0.0006208296],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001821951,0.0006391326,0.07429761,0.001131633,0.0009978422,0.0007514536,0.00158596,0.5054873,0.01279183,0.009583396,0.02927069,0.3616411],"study_design_scores_gemma":[0.0001545354,0.0007097787,0.02668766,0.0002530417,0.0001452764,0.000593266,0.0004361356,0.9095069,0.0211554,0.02978837,0.01044752,0.0001222477],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3484637,0.003275365,0.6069241,0.01112455,0.00106459,0.0007491077,0.004323264,0.01780783,0.006267658],"genre_scores_gemma":[0.8934546,0.0003586994,0.09953709,0.001277117,0.00017969,0.0003730987,0.002780408,0.0008335332,0.001205734],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9708067,"threshold_uncertainty_score":0.1543908,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08851988355023072,"score_gpt":0.2733285824830718,"score_spread":0.1848086989328411,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}