{"id":"W4415325631","doi":"10.1101/2025.10.17.25337471","title":"Artificial Intelligence-assisted reader evaluation in acute CT head interpretation (AI-REACT): a multireader multicase study","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"University of Oxford","keywords":"Diagnostic accuracy; Confidence interval; Head trauma; Computed tomography; Emergency department; Neuroimaging; Triage; Head (geology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04067538,0.001155163,0.00108195,0.002929637,0.0003316724,0.00224622,0.00147889,0.001219384,0.001342478],"category_scores_gemma":[0.1300982,0.0009247794,0.00251491,0.002628186,0.0008663445,0.002133744,0.001905682,0.0008464637,0.0007401246],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000723204,"about_ca_system_score_gemma":0.0005061289,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001041572,"about_ca_topic_score_gemma":0.00080536,"domain_scores_codex":[0.9542032,0.02858428,0.005016203,0.006305189,0.005307132,0.0005839719],"domain_scores_gemma":[0.8282591,0.0946243,0.04391449,0.01723553,0.01421691,0.001749713],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.005354346,0.0003259439,0.9710595,0.0006322691,0.004561755,0.0001043352,0.0007978698,0.0006593679,0.000464705,0.0001210145,0.0005138898,0.01540508],"study_design_scores_gemma":[0.0004654512,0.003344833,0.9815044,0.0002065815,0.00286957,0.00106921,0.0004350767,0.007388285,0.0008631156,0.0002037826,0.00154469,0.000105123],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.992077,0.003087372,0.003019849,0.0000746093,0.00004191873,0.0002975902,0.0007013893,0.00006686633,0.0006335029],"genre_scores_gemma":[0.9966601,0.0003482456,0.001797466,0.00007417334,0.00005240851,0.0001585293,0.0006733388,0.0000378187,0.0001979953],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04067538,"threshold_uncertainty_score":0.2151145,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07239348744651863,"score_gpt":0.4276278960154223,"score_spread":0.3552344085689036,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}