{"id":"W4393157213","doi":"10.1609/aaai.v38i20.30205","title":"MedAlign: A Clinician-Generated Dataset for Instruction Following with Electronic Medical Records","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Electronic Health Records Systems","field":"Health Professions","cited_by":46,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Institute of Biomedical Imaging and Bioengineering; National Institute of Arthritis and Musculoskeletal and Skin Diseases; National Institute of General Medical Sciences; National Heart, Lung, and Blood Institute","keywords":"Medical record; Electronic medical record; Computer science; Medical education; Psychology; Data science; Medicine; Information retrieval; Internet privacy; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003819413,0.0003220679,0.000544164,0.0002099649,0.000674666,0.00007570392,0.0009176279,0.0004080627,0.0003348764],"category_scores_gemma":[0.001622541,0.0002140593,0.0001613346,0.0009780602,0.0001811614,0.0002931393,0.0001282531,0.001555945,0.0001716826],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004497839,"about_ca_system_score_gemma":0.003561684,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004716561,"about_ca_topic_score_gemma":0.001463848,"domain_scores_codex":[0.9956642,0.0001384639,0.001470976,0.0007349314,0.0008542244,0.001137217],"domain_scores_gemma":[0.9975736,0.0007826363,0.0005082493,0.0003059249,0.0006028434,0.0002267197],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001167532,0.0001733213,0.001263017,0.002042371,0.0002753361,0.000003557489,0.002435751,0.000008576289,0.006852195,0.8231372,0.01587692,0.1467643],"study_design_scores_gemma":[0.001388064,0.008795928,0.0002266075,0.02885384,0.0007223706,0.00009962134,0.01743017,0.2250776,0.1386378,0.345803,0.2303626,0.002602413],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.93456,0.0005872458,0.009921543,0.02734244,0.009704086,0.007448589,0.0005454819,0.0005586664,0.00933196],"genre_scores_gemma":[0.9964036,0.0001826764,0.0003137584,0.0007782352,0.0008240629,0.0006474591,0.00006607479,0.00006368815,0.0007204966],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4773341,"threshold_uncertainty_score":0.8729085,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1508551813825793,"score_gpt":0.4613888072710266,"score_spread":0.3105336258884474,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}