{"id":"W6977765125","doi":"10.6084/m9.figshare.c.7612879.v1","title":"External validation of AI-based scoring systems in the ICU: a systematic review and meta-analysis","year":2025,"lang":"en","type":"other","venue":"Figshare","topic":"COVID-19 Digital Contact Tracing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Overfitting; Reliability (semiconductor); Receiver operating characteristic; External validity; Intensive care unit; Risk assessment; Intensive care","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0760923,0.00260466,0.01526213,0.009607514,0.0007918827,0.005413427,0.0031469,0.00233912,0.00308506],"category_scores_gemma":[0.1968315,0.001248223,0.02978178,0.01103721,0.001714804,0.00362677,0.002167444,0.002410459,0.0003946082],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003026948,"about_ca_system_score_gemma":0.004967767,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00483129,"about_ca_topic_score_gemma":0.01008444,"domain_scores_codex":[0.9327514,0.03421183,0.01988082,0.004850078,0.007367686,0.0009381493],"domain_scores_gemma":[0.7035102,0.2494879,0.02944072,0.006771038,0.01015758,0.0006326454],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"meta_analysis","study_design_scores_codex":[0.001225931,0.00002903242,0.01561536,0.5115233,0.4480773,0.0001113589,0.0001525686,0.000727312,0.0001460447,0.0003118038,0.001021223,0.02105878],"study_design_scores_gemma":[0.0005411837,0.0002516424,0.01520798,0.1638807,0.8134618,0.0001965391,0.0001252863,0.0006266124,0.0002864768,0.0009247232,0.004432506,0.00006463796],"study_design_candidate":"meta_analysis","study_design_consensus":null,"genre_codex":"review","genre_gemma":"other","genre_scores_codex":[0.00412015,0.9926187,0.001174853,0.0004292566,0.000158579,0.0003272717,0.0008301714,0.0000266962,0.0003142848],"genre_scores_gemma":[0.294044,0.6912272,0.006079855,0.002496404,0.0007159945,0.002380135,0.002620067,0.0001075361,0.0003288562],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.9239077,"threshold_uncertainty_score":0.4024193,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09521250061295855,"score_gpt":0.3238378408534722,"score_spread":0.2286253402405136,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}