{"id":"W4415905951","doi":"10.1111/dar.70065","title":"Machine Learning Algorithms to Predict Heavy Episodic Drinking in the United States Using Survey Data","year":2025,"lang":"en","type":"article","venue":"Drug and Alcohol Review","topic":"Substance Abuse Treatment and Outcomes","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Structural Genomics Consortium; Public Health Ontario; University of Toronto; Centre for Addiction and Mental Health","funders":"National Institute on Alcohol Abuse and Alcoholism; National Institutes of Health","keywords":"Discriminative model; Survey data collection; Test (biology); Predictive modelling; Model validation; Test data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007805002,0.0009883384,0.0008716127,0.002955291,0.0004795911,0.001148154,0.001277569,0.0008818784,0.001473599],"category_scores_gemma":[0.01997761,0.0004162738,0.001100826,0.002054703,0.0003300036,0.0009280947,0.0007680251,0.001494045,0.0004141345],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001392775,"about_ca_system_score_gemma":0.001380137,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01904464,"about_ca_topic_score_gemma":0.01172172,"domain_scores_codex":[0.9985116,0.0008779868,0.0001271248,0.0002399136,0.0001593719,0.00008391927],"domain_scores_gemma":[0.9879758,0.0100345,0.0008028513,0.0003004199,0.0007271164,0.0001592792],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001351557,0.0002632036,0.0815157,0.0001453581,0.0004294786,0.00004838449,0.00009268756,0.8482,0.0001204506,0.001329863,0.002288512,0.0654312],"study_design_scores_gemma":[0.00001273337,0.00004303002,0.004945679,0.0000369921,0.00002125348,0.00001167175,0.00003364975,0.9919647,0.00008481858,0.002606014,0.0002321613,0.000007231965],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7493861,0.002814006,0.2360932,0.002787512,0.0001576899,0.0004448137,0.003498105,0.0015713,0.003247238],"genre_scores_gemma":[0.9370568,0.0005749887,0.05843097,0.0002284379,0.0000751811,0.0003151118,0.00261331,0.00003146278,0.0006736427],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01904464,"threshold_uncertainty_score":0.04127729,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1059651283166314,"score_gpt":0.3778669965816435,"score_spread":0.2719018682650122,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}