{"id":"W4410201002","doi":"10.1007/s10278-025-01523-5","title":"Cross-Institutional Evaluation of Large Language Models for Radiology Diagnosis Extraction: A Prompt-Engineering Perspective","year":2025,"lang":"en","type":"article","venue":"Journal of Imaging Informatics in Medicine","topic":"Radiology practices and education","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Perspective (graphical); Extraction (chemistry); Computer science; Radiology; Medicine; Artificial intelligence; Chemistry; Chromatography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003838871,0.00008851423,0.0003586712,0.0005975222,0.0000456767,0.000007526804,0.00008072801,0.00005870567,0.00004473844],"category_scores_gemma":[0.004032169,0.00006886506,0.00006626393,0.0002326329,0.0001040184,0.0006149515,0.0000112389,0.0003144714,3.500055e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004760094,"about_ca_system_score_gemma":0.0005410099,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002174971,"about_ca_topic_score_gemma":0.00000216415,"domain_scores_codex":[0.9985892,0.0000373929,0.0008635325,0.0000562729,0.0003116715,0.0001419998],"domain_scores_gemma":[0.9976267,0.0004083081,0.000571367,0.0001127468,0.001237334,0.00004358178],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002965853,0.001983935,0.2500107,0.005071103,0.001520765,0.00007746734,0.1529795,0.4272261,0.004082716,0.07778213,0.01776788,0.05853191],"study_design_scores_gemma":[0.009251749,0.0003334872,0.05934094,0.001761767,0.0007119786,0.001152219,0.0236619,0.8990829,0.000469887,0.00273284,0.001400653,0.00009966904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8793607,0.005238948,0.102009,0.007663007,0.001394715,0.0006161242,0.000005252122,0.000009898763,0.003702363],"genre_scores_gemma":[0.9866034,0.000256357,0.0123744,0.000378547,0.000305687,0.00003314996,0.00001199299,0.000004884434,0.00003157253],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4718568,"threshold_uncertainty_score":0.4827174,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03598449043582892,"score_gpt":0.4389903356725702,"score_spread":0.4030058452367413,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}