{"id":"W4412487406","doi":"10.2214/ajr.25.33387","title":"General-Purpose and Reasoning Large Language Models for Automated Abdominal CT Protocoling","year":2025,"lang":"en","type":"article","venue":"American Journal of Roentgenology","topic":"Clinical practice guidelines implementation","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"Kingston Health Sciences Centre; University of Saskatchewan; Queen's University","funders":"","keywords":"Medicine; Natural language processing; Radiology; Medical physics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009327674,0.0001116381,0.0004967084,0.0002332107,0.00006781188,0.00001437126,0.00008393502,0.00002716581,0.00001839523],"category_scores_gemma":[0.001464034,0.00009547394,0.0001030846,0.0001985595,0.0001134996,0.0001326386,0.00003902406,0.0002341105,9.355825e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007810127,"about_ca_system_score_gemma":0.0002690533,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008460764,"about_ca_topic_score_gemma":0.00001701054,"domain_scores_codex":[0.9984297,0.0001308408,0.000884827,0.0001603147,0.0001270269,0.0002673189],"domain_scores_gemma":[0.9982846,0.0004099759,0.0007536386,0.0001397617,0.00030874,0.0001033208],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01479488,0.001200482,0.04623649,0.0004463966,0.002063999,0.001052257,0.002847821,0.002418348,0.05853089,0.004609155,0.01271835,0.8530809],"study_design_scores_gemma":[0.05193827,0.01998033,0.02581907,0.001360004,0.003724877,0.0115972,0.03650583,0.7941563,0.008431965,0.003434952,0.04225991,0.0007912968],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9211118,0.0003015202,0.07198916,0.004558485,0.0001360797,0.0017242,0.000009789239,0.00004385907,0.000125091],"genre_scores_gemma":[0.906778,0.00006493464,0.09004639,0.002688822,0.0001722008,0.000118185,0.000009798915,0.00001771095,0.0001039514],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8522896,"threshold_uncertainty_score":0.3893315,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05553583362942555,"score_gpt":0.4690247607528358,"score_spread":0.4134889271234103,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}