{"id":"W6929234836","doi":"10.48448/jame-c014","title":"ChatGPT for Suicide Risk Assessment on Social Media: Quantitative Evaluation of Model Performance, Potentials and Limitations","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; University of Ottawa","funders":"","keywords":"Context (archaeology); Task (project management); Hyperparameter; Suicide Risk; Mental health; Risk assessment; Poison control","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.007148374,0.000339509,0.0005047952,0.001788824,0.0005582467,0.00009249923,0.0004169277,0.0002096881,0.00004326293],"category_scores_gemma":[0.003406747,0.0003161413,0.00008862087,0.001250061,0.001398143,0.0003864268,0.0001133941,0.0002555957,0.0001086675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00046571,"about_ca_system_score_gemma":0.001899662,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009136502,"about_ca_topic_score_gemma":0.003657028,"domain_scores_codex":[0.9954171,0.0001897631,0.0005272329,0.0007667706,0.002614892,0.0004842237],"domain_scores_gemma":[0.9960304,0.001087149,0.001083168,0.0003231702,0.001366467,0.0001095875],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003455433,0.001586718,0.0008919458,0.0005958603,0.0009261424,0.000001833638,0.01268491,0.2697465,0.01168973,0.1540902,0.1185699,0.4288707],"study_design_scores_gemma":[0.001234259,0.0002297568,0.004736803,0.0002344143,0.0005204817,3.935321e-7,0.00100904,0.9838918,0.000224691,0.007357903,0.0002144195,0.0003460212],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.3195108,0.001977585,0.164371,0.002413142,0.007735263,0.02767928,0.03103405,0.004470869,0.4408081],"genre_scores_gemma":[0.9210136,0.0009420019,0.06968804,0.00004306856,0.0003863968,0.0006145224,0.0004320205,0.001289982,0.005590369],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7141453,"threshold_uncertainty_score":0.9999291,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2776658240438387,"score_gpt":0.4443548283367687,"score_spread":0.16668900429293,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}