{"id":"W7123307329","doi":"10.13026/t075-g517","title":"Lunguage: A Benchmark for Structured and Sequential Chest X-ray Interpretation","year":2025,"lang":"","type":"dataset","venue":"PhysioNet","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Benchmark (surveying); Pairwise comparison; SNOMED CT; Semantics (computer science); Interpretation (philosophy); Gold standard (test); Resource (disambiguation)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["metaepi_narrow"],"category_scores_codex":[0.0005751174,0.001697009,0.001776141,0.00089024,0.0007145624,0.0007040973,0.001149016,0.0009597086,0.001291199],"category_scores_gemma":[0.0008585011,0.001823556,0.00055092,0.001004922,0.0005267992,0.0007148052,0.001099349,0.001173307,0.0001628353],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004179363,"about_ca_system_score_gemma":0.0008074968,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002663034,"about_ca_topic_score_gemma":0.0002618098,"domain_scores_codex":[0.9934158,0.0005351611,0.001467617,0.002494971,0.0009304375,0.001156017],"domain_scores_gemma":[0.9950029,0.0006458939,0.001416455,0.001957108,0.0005658505,0.0004117773],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002982594,0.000329518,0.000002500306,0.005131118,0.001026826,0.0000125936,0.0009161982,0.000408486,0.03697341,0.00007491051,0.9471328,0.005009097],"study_design_scores_gemma":[0.006744603,0.0008276976,0.001094995,0.003218818,0.004994784,0.00001842588,0.000482293,0.0351202,0.002228695,0.002659751,0.9396658,0.002943913],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.002220735,0.001773305,0.001126225,0.0000801385,0.002878512,0.004567079,0.9871877,0.0001018416,0.00006448022],"genre_scores_gemma":[0.08339707,0.0002817335,0.001775307,0.0002433343,0.001230653,0.000679754,0.9119197,0.0001529333,0.0003195427],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.08117633,"threshold_uncertainty_score":0.9996217,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00878072575010086,"score_gpt":0.2949954740640469,"score_spread":0.2862147483139461,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}