{"id":"W4412607413","doi":"10.1200/cci-24-00317","title":"Extraction of Social Determinants of Health From Electronic Health Records Using Natural Language Processing","year":2025,"lang":"en","type":"article","venue":"JCO Clinical Cancer Informatics","topic":"Food Security and Health in Diverse Populations","field":"Health Professions","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; University of British Columbia, Okanagan Campus; Kelowna General Hospital","funders":"","keywords":"Pipeline (software); Benchmark (surveying); Artificial intelligence; Computer science; Social determinants of health; Natural language processing; Machine learning; Social media; Medicine; Public health; World Wide Web; Nursing; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003309126,0.001430074,0.0007012936,0.006990956,0.0007692802,0.00179539,0.0009593282,0.000916016,0.002473495],"category_scores_gemma":[0.01660884,0.0004307886,0.00190142,0.004634154,0.0006424473,0.001833637,0.001816212,0.001386856,0.002048117],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001305174,"about_ca_system_score_gemma":0.005309011,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01284561,"about_ca_topic_score_gemma":0.01347873,"domain_scores_codex":[0.9973483,0.000689905,0.0005573979,0.0007752174,0.0005211383,0.0001080663],"domain_scores_gemma":[0.9816411,0.01324583,0.001623525,0.001123577,0.002201838,0.0001640628],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005287337,0.0005586602,0.05894682,0.007577589,0.0004155365,0.003351306,0.002749981,0.01844898,0.0291737,0.006875568,0.06050044,0.8108727],"study_design_scores_gemma":[0.0003908894,0.0006637927,0.1482586,0.002600242,0.001254689,0.00337052,0.006499572,0.3768356,0.07671536,0.0618851,0.3210645,0.0004612776],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1618264,0.007719098,0.5839638,0.008595063,0.0007098538,0.003918943,0.2043111,0.01874444,0.01021135],"genre_scores_gemma":[0.2155287,0.002301432,0.6022452,0.0008552583,0.0003440036,0.001349253,0.174691,0.0003641056,0.002321115],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01284561,"threshold_uncertainty_score":0.02554166,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2724807600125266,"score_gpt":0.6235776239800963,"score_spread":0.3510968639675697,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}