{"id":"W4415077466","doi":"10.1007/978-3-032-07502-4_3","title":"Mind the Evaluation Gap: Large Language Models for Structured Data Extraction from Radiology Reports","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"McMaster University","funders":"","keywords":"Benchmarking; Structured prediction; Process (computing); Information extraction; Topic model; Data extraction","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.002873068,0.000372244,0.0004112052,0.000373375,0.0003042592,0.0004280244,0.004616279,0.0003673341,0.00001568539],"category_scores_gemma":[0.0003671258,0.0002894041,0.00009000153,0.0002765377,0.0001574627,0.0009847635,0.001999272,0.0005990123,0.000001813554],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002674404,"about_ca_system_score_gemma":0.0009184552,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006582378,"about_ca_topic_score_gemma":0.0002863505,"domain_scores_codex":[0.9957315,0.00009542039,0.0006218489,0.00216796,0.0008946935,0.0004886165],"domain_scores_gemma":[0.9937766,0.0009871855,0.0004685537,0.004422273,0.0002774017,0.00006803311],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004237767,0.00000971079,0.000005792159,0.0000129987,0.00002087076,0.00003136873,0.001241886,0.1162823,0.0001631631,0.008316415,0.0001112154,0.8738001],"study_design_scores_gemma":[0.0001665393,0.00001801019,0.00001704826,0.00007475331,0.00002173993,0.00005300403,6.455186e-7,0.7268273,0.0002396194,0.2715516,0.0008179772,0.0002116869],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0001113394,0.001695507,0.9915722,0.000983333,0.003913511,0.00104005,0.00006009661,0.0000712919,0.0005526526],"genre_scores_gemma":[0.1830832,0.00002558701,0.8144891,0.001068995,0.0008243051,0.00003283999,0.00022255,0.00001908483,0.0002343248],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8735884,"threshold_uncertainty_score":0.9999558,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06755437887015867,"score_gpt":0.3353343658884651,"score_spread":0.2677799870183064,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}