{"id":"W4386588264","doi":"10.1186/s44247-023-00035-y","title":"Discovering social determinants of health from case reports using natural language processing: algorithmic development and validation","year":2023,"lang":"en","type":"article","venue":"BMC Digital Health","topic":"Food Security and Health in Diverse Populations","field":"Health Professions","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; York University; Public Health Ontario; University of Toronto","funders":"Institute of Health Services and Policy Research; Canadian Institutes of Health Research","keywords":"Computer science; Natural language processing; Artificial intelligence; Benchmark (surveying); Social determinants of health; Social media; Annotation; Information extraction; Set (abstract data type); Key (lock); Information retrieval; Health care; Data science; Machine learning; World Wide Web; Political science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.0009211313,0.000168134,0.0004578331,0.000187318,0.001966317,0.00003974089,0.00005252368,0.0001077222,0.000008513294],"category_scores_gemma":[0.0001339174,0.0001686096,0.00003959302,0.0003740849,0.00005540685,0.000520104,0.0001676526,0.000291816,0.00001253909],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003959898,"about_ca_system_score_gemma":0.002433003,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004590537,"about_ca_topic_score_gemma":0.002732075,"domain_scores_codex":[0.9971699,0.0002086238,0.001332287,0.0003553952,0.0003002517,0.0006335578],"domain_scores_gemma":[0.9983827,0.0001590521,0.00102988,0.000156595,0.00007852801,0.0001932747],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00008271416,0.000170396,0.4609048,0.009183349,0.00002458529,0.0005704201,0.2157837,0.00004153369,0.000009121406,0.0001266174,0.0005286759,0.3125741],"study_design_scores_gemma":[0.00292224,0.0003106789,0.7570953,0.005030141,0.0000339886,0.0008194519,0.1846693,0.04341486,0.00007969485,0.001088301,0.003381092,0.001154974],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.996837,0.0008464347,0.0003145094,0.0002604093,0.0005050079,0.0008149958,0.000219645,0.0001440956,0.00005790697],"genre_scores_gemma":[0.9965948,0.00001308214,0.002484785,0.0001746292,0.0001712676,0.00002903078,0.0003873879,0.00002848666,0.0001165431],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3114192,"threshold_uncertainty_score":0.999333,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2337726519686488,"score_gpt":0.4902910702304957,"score_spread":0.256518418261847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}