{"id":"W6910420160","doi":"10.48448/fexf-9j94","title":"EHR-SeqSQL : A Sequential Text-to-SQL Dataset For Interactively Exploring Electronic Health Records","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Bridge (graph theory); Benchmark (surveying); Generalization; Electronic health record; Set (abstract data type); Health records; Data set; Test set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.003322799,0.0008759594,0.0008997213,0.003513889,0.0003841251,0.0006940249,0.002376642,0.0001979502,0.001099999],"category_scores_gemma":[0.0005388716,0.0008487882,0.0001731365,0.00321216,0.0007363363,0.0009238017,0.0009916913,0.001118829,0.01642364],"about_ca_system_candidate":true,"about_ca_system_consensus":true,"about_ca_system_score_codex":0.003894695,"about_ca_system_score_gemma":0.006048282,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002648218,"about_ca_topic_score_gemma":0.01066861,"domain_scores_codex":[0.992281,0.0001373318,0.0008371539,0.002690091,0.001412336,0.002642139],"domain_scores_gemma":[0.9968238,0.0001313875,0.0006365973,0.001463481,0.0001887454,0.0007559269],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008981424,0.0001276573,0.000002567176,0.0002535549,0.0001098795,0.000009333965,0.0003528341,0.00002616082,0.001766252,0.003574674,0.9765821,0.01710521],"study_design_scores_gemma":[0.0004724171,0.0007843241,0.000002990218,0.00096229,0.0000778969,0.00003671404,0.0003422145,0.002161161,0.0005162355,0.001414729,0.9923444,0.0008846723],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"other","genre_scores_codex":[0.001267094,0.01086857,0.1640203,0.02205745,0.04738996,0.03446512,0.451342,0.01213598,0.2564535],"genre_scores_gemma":[0.02823798,0.001003422,0.1061833,0.007947511,0.01355571,0.00546114,0.04028532,0.01457088,0.7827548],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.5263012,"threshold_uncertainty_score":0.9999292,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1141853089604736,"score_gpt":0.400927725526733,"score_spread":0.2867424165662594,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}