{"id":"W4410556235","doi":"10.2196/69709","title":"A Comparison of Responses from Human Therapists and Large Language Model–Based Chatbots to Assess Therapeutic Communication: Mixed Methods Study","year":2025,"lang":"en","type":"article","venue":"JMIR Mental Health","topic":"Digital Mental Health Interventions","field":"Psychology","cited_by":37,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Chatbot; Thematic analysis; Mental health; Psychological intervention; Psychology; Think aloud protocol; Intervention (counseling); Applied psychology; Medical education; Medicine; Qualitative research; Psychotherapist; Usability; Computer science; Psychiatry; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06020135,0.0007978753,0.001310306,0.00361315,0.002583002,0.002923561,0.001519058,0.001766976,0.001943223],"category_scores_gemma":[0.09382788,0.0007719153,0.001065449,0.002624266,0.002193852,0.002378069,0.003564385,0.00114172,0.0005884307],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002843364,"about_ca_system_score_gemma":0.002973479,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001143579,"about_ca_topic_score_gemma":0.002503056,"domain_scores_codex":[0.9186462,0.06096114,0.007242444,0.004405376,0.006945575,0.001799473],"domain_scores_gemma":[0.8853334,0.07587198,0.01373454,0.004737279,0.01832774,0.001995052],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.001567995,0.002953153,0.1978515,0.004766535,0.0004765452,0.001303781,0.6857675,0.0003309679,0.009187457,0.0012179,0.00190416,0.09267252],"study_design_scores_gemma":[0.0003884095,0.006923669,0.2801329,0.003672978,0.0003519331,0.001747184,0.6776174,0.00327121,0.006270423,0.001783623,0.01754581,0.0002945901],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9828799,0.0005440831,0.008933562,0.000261775,0.00006473308,0.00468696,0.0003658839,0.00003214112,0.002230969],"genre_scores_gemma":[0.9525896,0.0007039636,0.0190936,0.001297609,0.00009536401,0.02424728,0.0005561365,0.00006899485,0.001347343],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06020135,"threshold_uncertainty_score":0.318379,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1600705795782897,"score_gpt":0.6099041579763848,"score_spread":0.4498335783980951,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}