{"id":"W4405499503","doi":"10.1148/radiol.241711","title":"Methodological Challenges in Evaluating Large Language Models in Radiology","year":2024,"lang":"en","type":"letter","venue":"Radiology","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University","funders":"","keywords":"Medicine; Medical physics; Radiology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity"],"consensus_categories":["research_integrity"],"category_scores_codex":[0.005518354,0.0004810244,0.001999511,0.0009399119,0.00002987548,0.00001372705,0.0003425751,0.002115412,0.0002216911],"category_scores_gemma":[0.002750154,0.0003843056,0.0002532863,0.0002742362,0.0002276579,0.00004223017,0.00018463,0.008610948,0.00007564983],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003245095,"about_ca_system_score_gemma":0.0002060136,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009404733,"about_ca_topic_score_gemma":0.00003039236,"domain_scores_codex":[0.9937721,0.002769111,0.0008478379,0.001177829,0.0002949202,0.001138205],"domain_scores_gemma":[0.9967648,0.002385682,0.0001600639,0.0005515277,0.00003145288,0.0001064533],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003823752,0.0002643545,0.002672698,0.003643123,0.0006009073,0.08985948,0.008960476,0.001700993,0.001475561,0.006656572,0.7428365,0.140947],"study_design_scores_gemma":[0.006700872,0.001556104,0.003742857,0.002030254,0.000448161,0.01823889,0.0008440936,0.3235544,0.00001202022,0.03196817,0.6095365,0.001367747],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.02987473,0.09102664,0.001134859,0.8680218,0.001461767,0.0006924745,0.00001766993,0.0001871437,0.007582914],"genre_scores_gemma":[0.04113055,0.01780267,0.02262767,0.9024495,0.01159122,0.0003835613,0.001016372,0.0003235484,0.002674914],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.3218534,"threshold_uncertainty_score":0.9998609,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2032164414098568,"score_gpt":0.4412040421299666,"score_spread":0.2379876007201098,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}