{"id":"W4409250979","doi":"10.1016/j.compbiomed.2025.110161","title":"Integrating large language models with human expertise for disease detection in electronic health records","year":2025,"lang":"en","type":"article","venue":"Computers in Biology and Medicine","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Calgary Laboratory Services; Alberta Health Services; Libin Cardiovascular Institute of Alberta; University of Calgary","funders":"Canadian Institutes of Health Research","keywords":"Health records; Computer science; Electronic health record; Data science; Natural language processing; Artificial intelligence; Health care","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008266546,0.001671771,0.0008974667,0.002456609,0.0006976214,0.002401886,0.002081106,0.001318332,0.002442892],"category_scores_gemma":[0.02929969,0.001023557,0.00270767,0.001215081,0.0008868813,0.002723983,0.002676353,0.002663424,0.001570693],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002225179,"about_ca_system_score_gemma":0.002869001,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.022471,"about_ca_topic_score_gemma":0.03378531,"domain_scores_codex":[0.995595,0.002628472,0.0002426635,0.0009632158,0.0004159487,0.0001546162],"domain_scores_gemma":[0.9765742,0.02029817,0.0008259311,0.000944891,0.001100719,0.0002561391],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008709426,0.0009770797,0.0536205,0.000713832,0.001119429,0.001090667,0.001667006,0.5055923,0.006644087,0.01176594,0.01588931,0.400049],"study_design_scores_gemma":[0.00002891589,0.00003930503,0.001106165,0.00003302065,0.00006456493,0.00006837564,0.00005900153,0.986445,0.0006241817,0.01052017,0.0009887514,0.00002254207],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06141075,0.0009492706,0.9234655,0.002506572,0.000129735,0.0004183316,0.001992574,0.007157313,0.001969987],"genre_scores_gemma":[0.5380796,0.000579779,0.452658,0.001595517,0.0002485694,0.0006166891,0.003817659,0.0003413081,0.002062908],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.022471,"threshold_uncertainty_score":0.04468042,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01367169068403938,"score_gpt":0.3675827736706654,"score_spread":0.353911082986626,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}