{"id":"W4415077466","doi":"10.1007/978-3-032-07502-4_3","title":"Mind the Evaluation Gap: Large Language Models for Structured Data Extraction from Radiology Reports","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"McMaster University","funders":"","keywords":"Benchmarking; Structured prediction; Process (computing); Information extraction; Topic model; Data extraction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008505474,0.001057738,0.001175679,0.001592117,0.000575359,0.004001783,0.002201584,0.001362777,0.003139747],"category_scores_gemma":[0.02600479,0.0009171108,0.001647085,0.00195553,0.0008224835,0.007286891,0.002236129,0.003494605,0.002920585],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009493568,"about_ca_system_score_gemma":0.00205849,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002549344,"about_ca_topic_score_gemma":0.004445768,"domain_scores_codex":[0.9960169,0.002216566,0.0003490516,0.0005274883,0.0007847813,0.0001051389],"domain_scores_gemma":[0.9777089,0.01838822,0.0004974609,0.001885985,0.001260912,0.0002585558],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000603743,0.0003057907,0.003065042,0.0009430902,0.0003262999,0.0003063873,0.001268675,0.04684379,0.009448832,0.05464116,0.08360247,0.7986448],"study_design_scores_gemma":[0.00006560109,0.000113147,0.0009877797,0.0001965076,0.0001585774,0.0002289225,0.0003430066,0.8151516,0.009472625,0.1389835,0.03423402,0.00006468699],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008430503,0.001701464,0.9766815,0.00216474,0.0002132735,0.0001171797,0.002006171,0.006746438,0.001938778],"genre_scores_gemma":[0.1889692,0.002401227,0.7840906,0.00102112,0.0007160557,0.0004496568,0.01314043,0.003295996,0.005915798],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008505474,"threshold_uncertainty_score":0.04498178,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06755437887015867,"score_gpt":0.3353343658884651,"score_spread":0.2677799870183064,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}