{"id":"W4309321702","doi":"10.1148/ryai.220028","title":"Generalizability of Machine Learning Models: Quantitative Evaluation of Three Methodological Pitfalls","year":2022,"lang":"en","type":"article","venue":"Radiology Artificial Intelligence","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":131,"is_retracted":false,"has_abstract":true,"ca_institutions":"Jewish General Hospital; Montreal General Hospital; McGill University Health Centre; University of Calgary","funders":"National Institutes of Health; Fondation de l'Association des radiologistes du Québec","keywords":"Generalizability theory; Computer science; Machine learning; Artificial intelligence; Wilcoxon signed-rank test; Random forest; Overfitting; Feature selection; Artificial neural network; Data mining; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3208705,0.002491754,0.002086084,0.002955908,0.002266996,0.0052984,0.003629972,0.003134067,0.0012698],"category_scores_gemma":[0.6473677,0.001537572,0.003875897,0.002516906,0.008684951,0.004851755,0.006606765,0.004573065,0.00037841],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003633208,"about_ca_system_score_gemma":0.003023182,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003859251,"about_ca_topic_score_gemma":0.0027782,"domain_scores_codex":[0.7057648,0.2001459,0.03558791,0.02328777,0.03351967,0.001693964],"domain_scores_gemma":[0.2205174,0.6180629,0.0401021,0.09511484,0.02520907,0.0009937341],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004172945,0.0008602946,0.2892004,0.005696268,0.009029876,0.002380291,0.007706798,0.3283921,0.01713597,0.05069926,0.006425775,0.2782999],"study_design_scores_gemma":[0.001051912,0.004864623,0.09491668,0.002391726,0.00307436,0.00289917,0.002143362,0.6614853,0.03982792,0.1720321,0.01452635,0.0007865968],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1295117,0.002985863,0.8540498,0.004226772,0.000633181,0.00228516,0.0006656253,0.00186764,0.003774172],"genre_scores_gemma":[0.832231,0.0004880228,0.1611092,0.001699383,0.0003773791,0.002054787,0.0007873296,0.0006804059,0.0005724194],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6791295,"threshold_uncertainty_score":0.8374876,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.383065366064314,"score_gpt":0.4425682224172016,"score_spread":0.0595028563528876,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}