{"id":"W4410957636","doi":"10.1016/j.landig.2025.01.013","title":"Importance of sample size on the quality and utility of AI-based prediction models for healthcare","year":2025,"lang":"en","type":"review","venue":"The Lancet Digital Health","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":50,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"European Regional Development Fund; Medical Research Council; Vlaamse regering; South Asian Health Foundation; Vifor Pharma; KU Leuven; National Institute for Health and Care Research; Birmingham Biomedical Research Centre; Engineering and Physical Sciences Research Council; CSL Behring; UK Research and Innovation; Department of Health and Social Care; University Hospitals Birmingham NHS Foundation Trust; Cancer Research UK; Wellcome Trust; Fonds Wetenschappelijk Onderzoek","keywords":"Sample (material); Sample size determination; Health care; Computer science; Quality (philosophy); Agency (philosophy); Artificial intelligence; Outcome (game theory); Machine learning; Data science; Data mining; Risk analysis (engineering); Medicine; Statistics; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2200102,0.0009411756,0.005063442,0.002149256,0.0006864429,0.004491554,0.003399466,0.003523579,0.004081003],"category_scores_gemma":[0.4583192,0.000728722,0.003317686,0.002186587,0.004323571,0.005107777,0.002464433,0.005200181,0.0006690966],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002805096,"about_ca_system_score_gemma":0.006833657,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002734431,"about_ca_topic_score_gemma":0.002260576,"domain_scores_codex":[0.8201627,0.1399485,0.01589079,0.005855783,0.01741957,0.0007227335],"domain_scores_gemma":[0.3252246,0.6437727,0.009159226,0.008732642,0.01225137,0.0008593557],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001933721,0.0001305486,0.008840742,0.03163852,0.005141374,0.0001572106,0.0008308019,0.006638772,0.0003764956,0.04504755,0.01901948,0.8802447],"study_design_scores_gemma":[0.004691431,0.004319233,0.05552144,0.2111655,0.01886925,0.002339579,0.001457432,0.02144498,0.005243666,0.3128411,0.3614473,0.0006590383],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.01068276,0.8588459,0.06147173,0.05280879,0.005568095,0.001425181,0.001615861,0.0002429122,0.007338741],"genre_scores_gemma":[0.2904575,0.5584857,0.1037298,0.02778593,0.007009046,0.008721172,0.001786782,0.0002869732,0.001737039],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.7799898,"threshold_uncertainty_score":0.9618663,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5278624674459516,"score_gpt":0.5476276044938809,"score_spread":0.01976513704792937,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}