{"id":"W4409120023","doi":"10.2196/69534","title":"Is This Chatbot Safe and Evidence-Based? A Call for the Critical Evaluation of Generative AI Mental Health Chatbots","year":2025,"lang":"en","type":"article","venue":"Journal of Participatory Medicine","topic":"Digital Mental Health Interventions","field":"Psychology","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Chatbot; Mental health; Psychology; Internet privacy; Computer science; Medicine; World Wide Web; Psychiatry","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7056261,0.001728504,0.004393911,0.01141704,0.01423078,0.04001969,0.01393935,0.0223032,0.00653493],"category_scores_gemma":[0.8251096,0.001896912,0.00343196,0.005341703,0.06064418,0.0427773,0.02393847,0.02607462,0.001716457],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02598606,"about_ca_system_score_gemma":0.07916447,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00461189,"about_ca_topic_score_gemma":0.006668654,"domain_scores_codex":[0.2287541,0.6753581,0.04192737,0.007136131,0.04239877,0.00442555],"domain_scores_gemma":[0.03794256,0.8523329,0.02030309,0.02817135,0.05612224,0.005127801],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001052182,0.001054839,0.006484836,0.04093577,0.0008183529,0.0006984542,0.1259986,0.001254466,0.001142476,0.1800203,0.06374923,0.5767904],"study_design_scores_gemma":[0.0007751394,0.002049822,0.005605402,0.2648446,0.001125254,0.001114226,0.1524016,0.003645091,0.003902827,0.2499764,0.3138781,0.0006815937],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.0139612,0.05242962,0.08044432,0.8160765,0.01409388,0.00513127,0.0002027919,0.0006746957,0.0169858],"genre_scores_gemma":[0.3991474,0.02845763,0.3664585,0.1729307,0.005948475,0.02288826,0.0002884525,0.0009388885,0.002941725],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.2943739,"threshold_uncertainty_score":0.3630155,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3995995031340878,"score_gpt":0.5915934360208156,"score_spread":0.1919939328867278,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}