{"id":"W4410556235","doi":"10.2196/69709","title":"A Comparison of Responses from Human Therapists and Large Language Model–Based Chatbots to Assess Therapeutic Communication: Mixed Methods Study","year":2025,"lang":"en","type":"article","venue":"JMIR Mental Health","topic":"Digital Mental Health Interventions","field":"Psychology","cited_by":37,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Chatbot; Thematic analysis; Mental health; Psychological intervention; Psychology; Think aloud protocol; Intervention (counseling); Applied psychology; Medical education; Medicine; Qualitative research; Psychotherapist; Usability; Computer science; Psychiatry; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001338446,0.000191779,0.000501991,0.0002304956,0.0003258116,0.00003698776,0.0003539128,0.00006851021,0.0001169478],"category_scores_gemma":[0.000009808675,0.0001899937,0.00007363413,0.0002986147,0.00007085913,0.00007986544,0.0001870205,0.0002104888,0.00001341988],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000225167,"about_ca_system_score_gemma":0.00009106889,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001017155,"about_ca_topic_score_gemma":0.001540198,"domain_scores_codex":[0.9961742,0.002104894,0.0008036748,0.0003761913,0.0001961111,0.0003449402],"domain_scores_gemma":[0.9983511,0.0003742445,0.0002697979,0.0008041635,0.00003739147,0.0001632953],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.004483395,0.04767928,0.1387114,0.001003291,0.000823047,0.000006124091,0.1947007,0.00001023277,0.004288689,0.03860575,0.01057736,0.5591107],"study_design_scores_gemma":[0.008460848,0.006431811,0.8637789,0.001242298,0.0000553613,0.000002014182,0.1074543,0.002096159,0.004047415,0.003007077,0.002926398,0.0004973551],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9844134,0.003474465,0.003974964,0.003510593,0.0003254178,0.002540877,0.0004186825,0.00008232099,0.00125928],"genre_scores_gemma":[0.9907175,0.00000211578,0.005239367,0.002058313,0.00000895081,0.0005920193,0.0001226975,0.00002465716,0.001234364],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7250676,"threshold_uncertainty_score":0.7747719,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1600705795782897,"score_gpt":0.6099041579763848,"score_spread":0.4498335783980951,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}