{"id":"W4387569618","doi":"10.1101/2023.10.11.23296733","title":"Generalisability of AI-based scoring systems in the ICU: a systematic review and meta-analysis","year":2023,"lang":"en","type":"review","venue":"medRxiv","topic":"Sepsis Diagnosis and Treatment","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"European Commission; Alexander von Humboldt-Stiftung","keywords":"Overfitting; Intensive care unit; Reliability (semiconductor); Receiver operating characteristic; MEDLINE; Medicine; Intensive care; External validity; Meta-analysis; Computer science; Clinical Practice; Data mining; Artificial intelligence; Machine learning; Intensive care medicine; Statistics; Physical therapy; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","metaepi_broad"],"consensus_categories":[],"category_scores_codex":[0.06528269,0.003066415,0.01875667,0.01042169,0.0008067733,0.005568517,0.00350138,0.00285227,0.004043835],"category_scores_gemma":[0.1638121,0.001688124,0.04113998,0.01132269,0.001865828,0.003653561,0.002416846,0.002462162,0.0004640493],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003720761,"about_ca_system_score_gemma":0.005011176,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006429367,"about_ca_topic_score_gemma":0.01156993,"domain_scores_codex":[0.9422054,0.02593547,0.01933613,0.005801271,0.005723665,0.0009980216],"domain_scores_gemma":[0.8156391,0.1524912,0.02012611,0.005036331,0.006155228,0.0005520618],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"meta_analysis","study_design_gemma":"meta_analysis","study_design_scores_codex":[0.001014361,0.00001624081,0.009518887,0.437411,0.5392195,0.0001011287,0.0001100221,0.0004564499,0.0001167053,0.0002406206,0.0007510565,0.01104404],"study_design_scores_gemma":[0.0005374112,0.0001615466,0.0096564,0.08901665,0.8964025,0.0001187348,0.00008195515,0.0003339525,0.0001608397,0.0007633737,0.002728134,0.0000386338],"study_design_candidate":"meta_analysis","study_design_consensus":"meta_analysis","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.004038536,0.9918625,0.001360326,0.0004841857,0.0002378485,0.0004928168,0.001110229,0.00003620717,0.0003773847],"genre_scores_gemma":[0.3165969,0.66762,0.00539166,0.002792879,0.000810894,0.003665998,0.002533223,0.0001225683,0.000465884],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9812433,"threshold_uncertainty_score":0.345252,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4978238421310032,"score_gpt":0.4659701937695305,"score_spread":0.03185364836147275,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}