{"id":"W3133501805","doi":"10.1186/s13195-021-00879-4","title":"Data analysis with Shapley values for automatic subject selection in Alzheimer’s disease data sets using interpretable machine learning","year":2021,"lang":"en","type":"article","venue":"Alzheimer s Research & Therapy","topic":"Dementia and Cognitive Impairment Research","field":"Medicine","cited_by":64,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute of Biomedical Imaging and Bioengineering; Canadian Institutes of Health Research; National Institutes of Health; Genentech; IXICO; H. Lundbeck A/S; Servier; Eisai; Northern California Institute for Research and Education; BioClinica; F. Hoffmann-La Roche; University of Southern California; Biogen; U.S. Department of Defense; Meso Scale Diagnostics; Alzheimer's Disease Neuroimaging Initiative; Novartis Pharmaceuticals Corporation; Pfizer; Eli Lilly and Company; Bristol-Myers Squibb; National Institute on Aging; Alzheimer's Association; Foundation for the National Institutes of Health","keywords":"Overfitting; Artificial intelligence; Machine learning; Feature selection; Random forest; Computer science; Test set; Cross-validation; Neuroimaging; Data mining; Medicine; Artificial neural network","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00442942,0.0002868411,0.0005873294,0.001010117,0.000420964,0.0002970669,0.0008797291,0.00007888459,0.001155489],"category_scores_gemma":[0.0006594773,0.0002314337,0.0001160145,0.00343107,0.0002022477,0.0008146224,0.0008276968,0.0008043913,0.00001839385],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008488807,"about_ca_system_score_gemma":0.001388972,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001165666,"about_ca_topic_score_gemma":0.001060222,"domain_scores_codex":[0.9943922,0.001278599,0.0004399338,0.001338587,0.001506747,0.001043887],"domain_scores_gemma":[0.9963927,0.0007018238,0.00009016409,0.001586142,0.0008221057,0.0004070428],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005071934,0.00188215,0.7992978,0.0001243965,0.0266718,0.0003766442,0.0005645723,0.0003070338,0.006721294,0.00001938168,0.001207523,0.1577555],"study_design_scores_gemma":[0.003587341,0.0008835038,0.1103979,0.0002153771,0.003922902,0.00003207519,0.0004653837,0.873453,0.002698516,0.0001184689,0.003917518,0.000307993],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9217718,0.06187639,0.008927552,0.002098342,0.0000589712,0.003937599,0.0007397105,0.000176116,0.000413501],"genre_scores_gemma":[0.986088,0.001770411,0.005339346,0.000174263,0.0001014973,0.0001676514,0.006056492,0.00008138668,0.0002209324],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8731459,"threshold_uncertainty_score":0.9997576,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2688370233607937,"score_gpt":0.4751454934411039,"score_spread":0.2063084700803102,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}