{"id":"W4393157213","doi":"10.1609/aaai.v38i20.30205","title":"MedAlign: A Clinician-Generated Dataset for Instruction Following with Electronic Medical Records","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Electronic Health Records Systems","field":"Health Professions","cited_by":46,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Institute of Biomedical Imaging and Bioengineering; National Institute of Arthritis and Musculoskeletal and Skin Diseases; National Institute of General Medical Sciences; National Heart, Lung, and Blood Institute","keywords":"Medical record; Electronic medical record; Computer science; Medical education; Psychology; Data science; Medicine; Information retrieval; Internet privacy; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002449076,0.001282921,0.0005883012,0.002301291,0.0008890426,0.0009414214,0.002098054,0.002894841,0.01184423],"category_scores_gemma":[0.02441013,0.0004309597,0.001020193,0.001904428,0.0005059198,0.001318032,0.001929697,0.001996051,0.01352063],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001453712,"about_ca_system_score_gemma":0.002873401,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01236377,"about_ca_topic_score_gemma":0.02634442,"domain_scores_codex":[0.9969071,0.001247539,0.0004916948,0.0007027212,0.0005036004,0.0001474367],"domain_scores_gemma":[0.9876249,0.006718724,0.0008098228,0.002039019,0.002156126,0.0006514066],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001458376,0.00108399,0.02902923,0.003237166,0.000204072,0.001517931,0.001385196,0.006518503,0.00674183,0.001842608,0.8448171,0.102164],"study_design_scores_gemma":[0.002130506,0.001491361,0.07795218,0.0008031409,0.0002422151,0.00444488,0.002019373,0.06699869,0.02927484,0.006810811,0.8074213,0.0004106493],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.1037489,0.001450534,0.01588873,0.003502182,0.0005956339,0.001807713,0.8485671,0.01794612,0.006493128],"genre_scores_gemma":[0.05502133,0.0002337284,0.02884622,0.00092427,0.00009280878,0.001281924,0.9106981,0.0004474563,0.002454143],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.01236377,"threshold_uncertainty_score":0.03962284,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1508551813825793,"score_gpt":0.4613888072710266,"score_spread":0.3105336258884474,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}