{"id":"W3183267169","doi":"10.1093/gigascience/giab055","title":"Preventing dataset shift from breaking machine-learning biomarkers","year":2021,"lang":"en","type":"review","venue":"GigaScience","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":92,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"National Institute of Biomedical Imaging and Bioengineering; National Institute of Mental Health; National Institutes of Health; Agence Nationale de la Recherche","keywords":"Biomarker; Machine learning; Artificial intelligence; Computer science; Biomarker discovery; Population; Medicine; Biology; Proteomics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01187087,0.00100701,0.001714753,0.003899182,0.0004910174,0.002682497,0.00254527,0.001970655,0.002452466],"category_scores_gemma":[0.03270656,0.0005462021,0.001866188,0.002852782,0.001507987,0.003018317,0.002042259,0.002470528,0.001987347],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001151849,"about_ca_system_score_gemma":0.003074239,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001582566,"about_ca_topic_score_gemma":0.001966936,"domain_scores_codex":[0.9950276,0.001500658,0.0007344184,0.0008494496,0.00169704,0.0001908147],"domain_scores_gemma":[0.9775498,0.01562689,0.002311859,0.001352384,0.002932959,0.0002260942],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001410218,0.00005405693,0.001921748,0.01361516,0.0004172842,0.0001922626,0.0001935536,0.0006035443,0.001783363,0.009443633,0.01783969,0.9537947],"study_design_scores_gemma":[0.00008372434,0.0002985897,0.007409838,0.02145425,0.00117529,0.003269723,0.0002690509,0.001542829,0.01052338,0.02239597,0.9314322,0.0001451257],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.002647284,0.9607883,0.0228601,0.006329379,0.002044871,0.0001835447,0.0005866157,0.000382896,0.004176925],"genre_scores_gemma":[0.03459569,0.9184275,0.03141757,0.007421936,0.001995521,0.0005098928,0.001922974,0.00018983,0.003519191],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01187087,"threshold_uncertainty_score":0.06277984,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2725123400981636,"score_gpt":0.4901105873463265,"score_spread":0.2175982472481629,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}