{"id":"W6929234836","doi":"10.48448/jame-c014","title":"ChatGPT for Suicide Risk Assessment on Social Media: Quantitative Evaluation of Model Performance, Potentials and Limitations","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; University of Ottawa","funders":"","keywords":"Context (archaeology); Task (project management); Hyperparameter; Suicide Risk; Mental health; Risk assessment; Poison control","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005336139,0.002014159,0.001028595,0.001315707,0.0007057122,0.001448217,0.002382453,0.002038088,0.001908942],"category_scores_gemma":[0.0151691,0.0004054731,0.0009635919,0.0006906945,0.0007494375,0.002020075,0.001818296,0.002742073,0.001248817],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001578927,"about_ca_system_score_gemma":0.001691653,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02401705,"about_ca_topic_score_gemma":0.02580477,"domain_scores_codex":[0.9982129,0.0008542328,0.00007944523,0.0004953429,0.0002372842,0.000120792],"domain_scores_gemma":[0.9947923,0.003635349,0.0001661647,0.0005211826,0.0006454407,0.0002395866],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001778893,0.0009679003,0.05002257,0.0007816826,0.0006817229,0.0005355956,0.0007350417,0.6042557,0.006674787,0.003124261,0.02778707,0.3026548],"study_design_scores_gemma":[0.00002920066,0.0001254711,0.001974687,0.00002968787,0.00005270505,0.00009291898,0.0001102708,0.9934492,0.001622701,0.001454527,0.001036024,0.00002264551],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5623813,0.004030254,0.3877582,0.003496187,0.0009876298,0.00134334,0.006629191,0.02321958,0.01015422],"genre_scores_gemma":[0.8936402,0.0003974742,0.09634893,0.000599781,0.0001094004,0.0005758413,0.005423624,0.0003938295,0.002510927],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02401705,"threshold_uncertainty_score":0.04775453,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2776658240438387,"score_gpt":0.4443548283367687,"score_spread":0.16668900429293,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}