{"id":"W4401043227","doi":"10.18653/v1/2024.clinicalnlp-1.62","title":"Overview of the EHRSQL 2024 Shared Task on Reliable Text-to-SQL Modeling on Electronic Health Records","year":2024,"lang":"en","type":"article","venue":"","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; National Research Foundation; National Research Foundation of Korea; Strong","keywords":"Computer science; Task (project management); SQL; Health records; Electronic health record; World Wide Web; Database; Health care; Engineering; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03714678,0.004203205,0.003335066,0.00554832,0.003200193,0.006306787,0.007427135,0.004184125,0.01740756],"category_scores_gemma":[0.0578324,0.002331253,0.00429204,0.005620311,0.002116441,0.008687042,0.01560209,0.00554631,0.01720773],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004462074,"about_ca_system_score_gemma":0.01415261,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04148037,"about_ca_topic_score_gemma":0.03129868,"domain_scores_codex":[0.9619132,0.0167729,0.004708461,0.00555975,0.008941457,0.002104301],"domain_scores_gemma":[0.960813,0.01373594,0.001215047,0.008964375,0.01092354,0.004348009],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002073068,0.003104995,0.005891583,0.005347377,0.0006130813,0.0007437537,0.003521937,0.02422555,0.0171993,0.01364579,0.6208867,0.3027469],"study_design_scores_gemma":[0.001552134,0.001352355,0.008155108,0.001044435,0.0002413887,0.0007765397,0.002055609,0.228743,0.02017739,0.03495104,0.7003635,0.0005876095],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.02936526,0.004722659,0.6894893,0.01208822,0.001127808,0.01328909,0.111436,0.1177715,0.02071024],"genre_scores_gemma":[0.05310715,0.001078464,0.5875514,0.002314346,0.0003514984,0.007310245,0.3328767,0.008428604,0.006981717],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.04148037,"threshold_uncertainty_score":0.1964533,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1986647325953776,"score_gpt":0.4249938521769817,"score_spread":0.2263291195816041,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}