{"id":"W4403052388","doi":"10.1007/978-3-031-72117-5_12","title":"Confidence Intervals Uncovered: Are We Ready for Real-World Medical Imaging AI?","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":9,"is_retracted":false,"has_abstract":false,"ca_institutions":"Queen's University","funders":"European Commission; Agence Nationale de la Recherche; Nvidia","keywords":"Computer science; Confidence interval; Medical imaging; Artificial intelligence; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01705853,0.00102886,0.001629393,0.002078443,0.0008611993,0.007838163,0.004033037,0.003409417,0.01168395],"category_scores_gemma":[0.1813273,0.0008269094,0.001230025,0.002138725,0.006925758,0.01658765,0.003055306,0.0114558,0.001795803],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001986287,"about_ca_system_score_gemma":0.001972896,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002992899,"about_ca_topic_score_gemma":0.001688529,"domain_scores_codex":[0.9935966,0.002925745,0.0004522142,0.0008879122,0.001901497,0.0002360283],"domain_scores_gemma":[0.854196,0.1258446,0.003506686,0.007026806,0.007424748,0.002001136],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003013076,0.0001135336,0.003059372,0.001045926,0.0002402563,0.0004094903,0.0008451524,0.007169912,0.0007535661,0.3480018,0.06276437,0.5752954],"study_design_scores_gemma":[0.00003325344,0.00004631317,0.0009291301,0.0005423541,0.0000638769,0.000414604,0.0003270737,0.0228869,0.0004954172,0.9410589,0.03314526,0.00005688735],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01463677,0.07694013,0.6831436,0.1852773,0.004290591,0.00008894451,0.001000593,0.001771333,0.03285082],"genre_scores_gemma":[0.5177313,0.03741341,0.3980106,0.02236598,0.00942699,0.0002176136,0.00145028,0.001574083,0.01180974],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9829414,"threshold_uncertainty_score":0.09021527,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1296929749811029,"score_gpt":0.4330544743597912,"score_spread":0.3033614993786883,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}