{"id":"W4311617908","doi":"10.1101/2022.11.30.22282946","title":"Discovering Social Determinants of Health from Case Reports using Natural Language Processing: Algorithmic Development and Validation","year":2022,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Public Health Ontario; University of Toronto","funders":"Institute of Health Services and Policy Research; Canadian Institutes of Health Research","keywords":"Computer science; Benchmark (surveying); Natural language processing; Artificial intelligence; Social media; Social determinants of health; Annotation; Key (lock); Information extraction; Set (abstract data type); Information retrieval; Health care; Data science; Machine learning; World Wide Web; Political science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01524309,0.001094777,0.0006454711,0.005391206,0.000856857,0.001985332,0.002326004,0.001609897,0.001883052],"category_scores_gemma":[0.04196019,0.0005029545,0.001462946,0.002079142,0.001357003,0.001791441,0.001992529,0.001615021,0.0007679196],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001130762,"about_ca_system_score_gemma":0.002312697,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005189486,"about_ca_topic_score_gemma":0.006063996,"domain_scores_codex":[0.991757,0.004875994,0.0008837456,0.001506365,0.0008112029,0.0001657393],"domain_scores_gemma":[0.918048,0.07404894,0.001816204,0.002877149,0.002927222,0.0002825533],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005735337,0.001369235,0.055159,0.001659904,0.0005601807,0.001504533,0.001796999,0.223627,0.006427783,0.008263349,0.01264047,0.686418],"study_design_scores_gemma":[0.00008127803,0.00008188894,0.00625639,0.0001321268,0.00008216971,0.0003458732,0.0003840271,0.9758427,0.003211798,0.01089624,0.002661613,0.00002395719],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1685836,0.001427852,0.8164659,0.001574647,0.0001081895,0.001953087,0.003422017,0.004462901,0.002001811],"genre_scores_gemma":[0.3164129,0.0003967577,0.673245,0.0001971025,0.0001153402,0.001291626,0.007749063,0.0001106062,0.000481563],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01524309,"threshold_uncertainty_score":0.08061415,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05517938981622243,"score_gpt":0.335244282657424,"score_spread":0.2800648928412016,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}