{"id":"W4410185915","doi":"10.2196/63272","title":"Improving Suicidal Ideation Detection in Social Media Posts: Topic Modeling and Synthetic Data Augmentation Approach","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Mental Health via Writing","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; University of Ottawa","keywords":"Preprint; Ideation; Social media; Suicidal ideation; Psychology; Data science; Sociology; Computer science; Suicide prevention; Poison control; World Wide Web; Medicine; Cognitive science; Medical emergency","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002529655,0.00009227588,0.0001398212,0.0005542515,0.0004079284,0.00007146469,0.0002156752,0.0001214967,0.00003266954],"category_scores_gemma":[0.0001874516,0.00009591591,0.00001313244,0.0005492404,0.00007499758,0.000594467,0.0002866054,0.0005431909,0.00001768568],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004047686,"about_ca_system_score_gemma":0.00006032225,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001003089,"about_ca_topic_score_gemma":0.0003887403,"domain_scores_codex":[0.9978986,0.0005881489,0.0003815181,0.000353549,0.0003509906,0.0004271787],"domain_scores_gemma":[0.99915,0.0003933072,0.000063508,0.0002422992,0.0001044329,0.00004639242],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004708826,0.0003383805,0.002053076,0.001164151,0.00003409908,0.000004690669,0.07078012,0.0000268427,0.003188561,0.0120649,0.0002097293,0.9096646],"study_design_scores_gemma":[0.002183706,0.0001358433,0.04232442,0.0001523054,0.000008828412,0.000009938933,0.09320083,0.8571799,0.0007018652,0.003881281,0.00002772776,0.0001933157],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9711792,0.0001524191,0.01709981,0.0002310028,0.0001926454,0.0008267144,0.00002099125,0.00003401478,0.01026317],"genre_scores_gemma":[0.9991683,0.000006967809,0.0002297949,0.00003078284,0.00009669099,0.0002811702,0.0001176328,0.000009853153,0.00005884182],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9094713,"threshold_uncertainty_score":0.3911337,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1686070679126437,"score_gpt":0.4998295271952565,"score_spread":0.3312224592826128,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}