{"id":"W4406971480","doi":"10.2196/68618","title":"Automated Radiology Report Labeling in Chest X-Ray Pathologies: Development and Evaluation of a Large Language Model Framework","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Computer science; Medicine; Radiology; Medical physics; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003828595,0.001388132,0.000712185,0.001549953,0.000438013,0.001419615,0.002523577,0.001627385,0.001926124],"category_scores_gemma":[0.006729574,0.0005376317,0.001843391,0.0007554899,0.0005232806,0.001806695,0.001304358,0.002279036,0.001688065],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002001127,"about_ca_system_score_gemma":0.002784854,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01828969,"about_ca_topic_score_gemma":0.02299227,"domain_scores_codex":[0.9987518,0.000412745,0.00008330029,0.0004142419,0.0002558418,0.00008203073],"domain_scores_gemma":[0.9961804,0.002455209,0.0001881815,0.0003438215,0.0006558864,0.0001764121],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006072649,0.0006266838,0.009110886,0.0003645409,0.0003395394,0.0005069507,0.0003226646,0.390555,0.01431691,0.003576454,0.01540365,0.5642695],"study_design_scores_gemma":[0.00001790671,0.00005315681,0.0004532299,0.00001722015,0.00003399172,0.00008552481,0.00003309056,0.994373,0.002868194,0.0009641804,0.001087125,0.00001341663],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1050422,0.001749283,0.8587008,0.001192233,0.0002643206,0.0004974135,0.002129392,0.0279641,0.002460225],"genre_scores_gemma":[0.4446985,0.000833039,0.5396876,0.0007104231,0.0001403444,0.0004486235,0.008481453,0.0008913435,0.004108703],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01828969,"threshold_uncertainty_score":0.03636646,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1044424482632265,"score_gpt":0.4812142868328491,"score_spread":0.3767718385696227,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}