{"id":"W3023773746","doi":"10.1016/j.media.2020.101714","title":"The reliability of a deep learning model in clinical out-of-distribution MRI data: A multicohort study","year":2020,"lang":"en","type":"article","venue":"Medical Image Analysis","topic":"MRI in cancer diagnosis","field":"Medicine","cited_by":171,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute on Aging; Canadian Institutes of Health Research; National Institutes of Health; IXICO; H. Lundbeck A/S; Servier; Karolinska Institutet; Alzheimerfonden; Swedish Brain Power; Hjärnfonden; Vetenskapsrådet; Eisai; Stiftelsen Olle Engkvist Byggmästare; Genentech; Stiftelsen för Strategisk Forskning; Center for Innovative Medicine; Northern California Institute for Research and Education; Stockholms Läns Landsting; DoD Alzheimer's Disease Neuroimaging Initiative; Pfizer; Biogen; BioClinica; Nvidia; University of Southern California; U.S. Department of Defense; Eli Lilly and Company; Bristol-Myers Squibb; Meso Scale Diagnostics; Alzheimer's Disease Neuroimaging Initiative; Novartis Pharmaceuticals Corporation; Alzheimer's Association; Åke Wiberg Stiftelse","keywords":"Artificial intelligence; Computer science; Reliability (semiconductor); Deep learning; Convolutional neural network; Protocol (science); Machine learning; Neuroradiologist; Neuroimaging; Medical physics; Pattern recognition (psychology); Magnetic resonance imaging; Medicine; Pathology; Radiology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07475618,0.001468105,0.001689037,0.001760488,0.0008375671,0.002312849,0.002731606,0.002442658,0.001258987],"category_scores_gemma":[0.1413786,0.0008652507,0.002313621,0.00106435,0.00261741,0.0028069,0.00268863,0.002116299,0.001006533],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008617961,"about_ca_system_score_gemma":0.0007737483,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003451012,"about_ca_topic_score_gemma":0.001654593,"domain_scores_codex":[0.9722043,0.01655379,0.001970836,0.00663403,0.001778652,0.0008585097],"domain_scores_gemma":[0.8079221,0.1199462,0.01385427,0.04041617,0.01487632,0.002984909],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.008699098,0.0008811336,0.9157873,0.0003113235,0.003836321,0.0005456995,0.001656053,0.02946939,0.002269337,0.0006661466,0.002750056,0.03312821],"study_design_scores_gemma":[0.0005786474,0.005703796,0.5387443,0.0004113867,0.002985762,0.003333946,0.003046007,0.4227481,0.008125905,0.00861022,0.00528069,0.0004312245],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9855222,0.001192221,0.0107859,0.0003695794,0.0001425901,0.0001132565,0.001252018,0.0001290246,0.0004932077],"genre_scores_gemma":[0.9956328,0.0001264772,0.001781978,0.000149183,0.00006075573,0.00005356201,0.001920851,0.00008960715,0.0001846965],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.07475618,"threshold_uncertainty_score":0.3953532,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07275462749978208,"score_gpt":0.4260945749124501,"score_spread":0.353339947412668,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}