{"id":"W4412487406","doi":"10.2214/ajr.25.33387","title":"General-Purpose and Reasoning Large Language Models for Automated Abdominal CT Protocoling","year":2025,"lang":"en","type":"article","venue":"American Journal of Roentgenology","topic":"Clinical practice guidelines implementation","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"Kingston Health Sciences Centre; University of Saskatchewan; Queen's University","funders":"","keywords":"Medicine; Natural language processing; Radiology; Medical physics; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006714814,0.001838397,0.001153526,0.002326868,0.001275876,0.006243584,0.004973625,0.002446949,0.01000192],"category_scores_gemma":[0.03785659,0.002317842,0.005262062,0.001475886,0.001207484,0.006030543,0.004143695,0.003716317,0.002804936],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002787575,"about_ca_system_score_gemma":0.00521815,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0187036,"about_ca_topic_score_gemma":0.04559579,"domain_scores_codex":[0.9942514,0.001855378,0.0009706495,0.0008290304,0.001748387,0.0003452976],"domain_scores_gemma":[0.9769791,0.01583918,0.001196774,0.003181338,0.002398946,0.0004046512],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001139323,0.001086671,0.01859318,0.003792839,0.001061435,0.002615334,0.003301316,0.3882758,0.01585874,0.1121251,0.0806432,0.371507],"study_design_scores_gemma":[0.0001348978,0.000057376,0.00056433,0.0002237213,0.0003070444,0.0003134965,0.0002565711,0.906319,0.01037683,0.05538033,0.02598819,0.00007813202],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01098835,0.0002189988,0.9340485,0.001222173,0.00007899107,0.0008336323,0.004444193,0.04587331,0.002291875],"genre_scores_gemma":[0.1162082,0.0002217968,0.87132,0.000510604,0.00004997417,0.000442429,0.007384848,0.002313397,0.001548771],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0187036,"threshold_uncertainty_score":0.03718948,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05553583362942555,"score_gpt":0.4690247607528358,"score_spread":0.4134889271234103,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}