{"id":"W7123307329","doi":"10.13026/t075-g517","title":"Lunguage: A Benchmark for Structured and Sequential Chest X-ray Interpretation","year":2025,"lang":"","type":"dataset","venue":"PhysioNet","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Benchmark (surveying); Pairwise comparison; SNOMED CT; Semantics (computer science); Interpretation (philosophy); Gold standard (test); Resource (disambiguation)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003200298,0.002766245,0.0008906777,0.005979548,0.000872496,0.002722296,0.004385602,0.002940119,0.005644734],"category_scores_gemma":[0.01733526,0.0005725662,0.001920088,0.004333711,0.000847208,0.002227815,0.002983437,0.001494597,0.005407282],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002534219,"about_ca_system_score_gemma":0.00328394,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02176316,"about_ca_topic_score_gemma":0.03813736,"domain_scores_codex":[0.9954851,0.001174574,0.0007936307,0.001294713,0.001012526,0.0002395149],"domain_scores_gemma":[0.9918047,0.003996614,0.0008875327,0.001478537,0.001296945,0.0005356196],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00137687,0.0007629853,0.03499479,0.005911021,0.0007488492,0.00158701,0.0006691951,0.02225247,0.005237401,0.005015464,0.8140541,0.1073899],"study_design_scores_gemma":[0.001641498,0.0007096727,0.06760655,0.002318853,0.0005467565,0.005624061,0.001760341,0.1345639,0.01861577,0.01500992,0.751246,0.0003566325],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.05247109,0.006641935,0.0151065,0.002011987,0.0002908153,0.000711191,0.8902789,0.02445606,0.008031536],"genre_scores_gemma":[0.03347931,0.0006160116,0.02312334,0.0003548026,0.0000519161,0.0003536342,0.9407431,0.0004195973,0.0008582001],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02176316,"threshold_uncertainty_score":0.04327291,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00878072575010086,"score_gpt":0.2949954740640469,"score_spread":0.2862147483139461,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}