{"id":"W4293918660","doi":"10.1101/2022.08.30.22279318","title":"Evaluating Progress in Automatic Chest X-Ray Radiology Report Generation","year":2022,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Topic Modeling","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"National Institute of Biomedical Imaging and Bioengineering; National Institutes of Health","keywords":"Workflow; Computer science; Metric (unit); Rank (graph theory); Quality (philosophy); Artificial intelligence; Medical imaging; Identification (biology); Contrast (vision); Medical physics; Information retrieval; Machine learning; Natural language processing; Data science; Radiology; Medicine; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05503156,0.002305701,0.001458792,0.008304799,0.00080575,0.00475189,0.003783455,0.002289626,0.001435132],"category_scores_gemma":[0.1719634,0.0006334435,0.001048291,0.003498847,0.001062257,0.003506466,0.002888364,0.001665673,0.001217336],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002270832,"about_ca_system_score_gemma":0.002380548,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005463732,"about_ca_topic_score_gemma":0.003540683,"domain_scores_codex":[0.942482,0.03560451,0.004863701,0.004753413,0.01133858,0.0009577306],"domain_scores_gemma":[0.7460912,0.1728144,0.01683211,0.02249875,0.03777398,0.003989638],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002989241,0.001849358,0.08658924,0.002127085,0.0009067888,0.0003046712,0.001739701,0.1753439,0.01633652,0.004025641,0.02612022,0.6816676],"study_design_scores_gemma":[0.0004398754,0.002629556,0.03311029,0.0002104568,0.0002871603,0.0004472501,0.0007691951,0.8904234,0.05634237,0.003770377,0.01133866,0.0002313069],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7436329,0.00883617,0.193468,0.002371699,0.0007576866,0.001035791,0.004356603,0.03940544,0.00613567],"genre_scores_gemma":[0.823254,0.0006223581,0.1635846,0.0002098242,0.0001813797,0.0002251431,0.01014121,0.00071866,0.001062843],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05503156,"threshold_uncertainty_score":0.2910382,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1107970950486012,"score_gpt":0.3653045299124789,"score_spread":0.2545074348638777,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}