{"id":"W4401943674","doi":"10.1109/icdh62654.2024.00031","title":"Enhancing Large Language Models with Human Expertise for Disease Detection in Electronic Health Records","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"Canadian Institutes of Health Research","keywords":"Computer science; Health records; Electronic health record; Human disease; Disease; Data science; Natural language processing; Health care; Medicine; Pathology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003207086,0.00008767001,0.00009625498,0.0001214245,0.00008124736,0.0001023256,0.000178956,0.00002071043,0.000005917226],"category_scores_gemma":[0.000004781162,0.00007243581,0.00003213875,0.0001940541,0.000003677983,0.000451637,0.00004554657,0.00010503,0.000002648296],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002670351,"about_ca_system_score_gemma":0.0001991286,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000404444,"about_ca_topic_score_gemma":0.008411319,"domain_scores_codex":[0.9989185,0.0000289369,0.000165328,0.0003744053,0.0001134396,0.0003994009],"domain_scores_gemma":[0.9995875,0.00002223415,0.000019812,0.0002727061,0.00001316493,0.00008455176],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000643418,0.0001790148,0.00009849244,0.0006054033,0.00003208145,0.00006094072,0.02142196,0.006979372,0.009109044,0.4949536,0.0001455679,0.4663502],"study_design_scores_gemma":[0.0002104103,0.0001092203,0.00001924206,0.00008250412,0.000001515843,0.000002311417,0.0001029485,0.9898065,0.002496573,0.00691271,0.0001518791,0.0001041399],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09432948,0.00118528,0.9033072,0.0004422636,0.00009053014,0.0002615569,0.000001021415,0.0002676398,0.000115041],"genre_scores_gemma":[0.9852512,0.00001288313,0.0138148,0.0002863588,0.00006319154,0.0001041676,0.000001996443,0.00001244898,0.000452882],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9828272,"threshold_uncertainty_score":0.4693713,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01501122271640687,"score_gpt":0.2914003633171336,"score_spread":0.2763891406007267,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}