{"id":"W2784112886","doi":"10.2196/publichealth.7726","title":"Detecting Novel and Emerging Drug Terms Using Natural Language Processing: A Social Media Corpus Study","year":2018,"lang":"en","type":"article","venue":"JMIR Public Health and Surveillance","topic":"Mental Health via Writing","field":"Psychology","cited_by":32,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute on Drug Abuse","keywords":"Social media; Public health; Data science; Drug; Field (mathematics); Internet privacy; Computer science; Sample (material); Natural (archaeology); Ask price; Illicit drug; Psychology; Medicine; Pharmacology; World Wide Web; Pathology; Geography; Business","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002223871,0.0002277314,0.0004137427,0.0001728462,0.001148554,0.0001442667,0.0001379833,0.00008313834,0.0000308153],"category_scores_gemma":[0.0002435853,0.0002183156,0.00002390014,0.0004245364,0.0001749334,0.0001939892,0.0001194649,0.0004054157,0.000005038379],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001259999,"about_ca_system_score_gemma":0.0002297327,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005811772,"about_ca_topic_score_gemma":0.001979513,"domain_scores_codex":[0.9971423,0.0003985221,0.0005657174,0.0005874061,0.0002848599,0.00102123],"domain_scores_gemma":[0.9986287,0.0002468788,0.0003643156,0.000177553,0.00009733883,0.000485232],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00008088302,0.0001962815,0.3477901,0.0003293375,0.00001604849,0.00001404192,0.1638534,1.574125e-8,0.0001351737,0.00005352546,0.0001570603,0.4873741],"study_design_scores_gemma":[0.00458386,0.0004209011,0.8224994,0.0000827717,0.00000213013,0.0002938681,0.1629232,0.004615437,0.000003429834,0.00001658405,0.003897654,0.0006608196],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.993569,0.002717542,0.0000761338,0.001356828,0.0007742253,0.0007480935,0.0000133849,0.0001472644,0.0005975674],"genre_scores_gemma":[0.9970341,0.000005138248,0.0003529163,0.001360981,0.001054288,0.00006403289,0.00001050985,0.00003963871,0.00007836884],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4867133,"threshold_uncertainty_score":0.8902652,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05855809624271563,"score_gpt":0.4035978985095184,"score_spread":0.3450398022668028,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}