{"id":"W4400904138","doi":"10.2196/57306","title":"Exploring the Efficacy of Large Language Models in Summarizing Mental Health Counseling Sessions: Benchmark Study","year":2024,"lang":"en","type":"article","venue":"JMIR Mental Health","topic":"Mental Health via Writing","field":"Psychology","cited_by":29,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Benchmarking; Mental health; Benchmark (surveying); Task (project management); Recall; Process (computing); Applied psychology; Psychology; Medical education; Artificial intelligence; Medicine; Psychotherapist; Cognitive psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007873961,0.002155254,0.0009124729,0.001948459,0.0005597518,0.001609872,0.001805585,0.001397416,0.001589805],"category_scores_gemma":[0.02705072,0.0004550807,0.001293303,0.001209036,0.0005682172,0.002394208,0.001413345,0.001859362,0.000813667],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001605713,"about_ca_system_score_gemma":0.001486416,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01071046,"about_ca_topic_score_gemma":0.01374023,"domain_scores_codex":[0.9952648,0.003061533,0.0003703918,0.000795282,0.0003609534,0.00014708],"domain_scores_gemma":[0.9764469,0.01936896,0.000621517,0.001511182,0.001675737,0.0003757857],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003214268,0.002500052,0.01563613,0.002761808,0.001201349,0.0005394949,0.001771073,0.3976156,0.01363987,0.001981707,0.01733734,0.5418013],"study_design_scores_gemma":[0.0003305114,0.00166199,0.004969361,0.0001275112,0.0003383874,0.0001452258,0.0006878497,0.9749591,0.01107978,0.001865578,0.003750203,0.00008452627],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8562158,0.006329682,0.113463,0.001377172,0.0004844846,0.001074946,0.007258328,0.01012664,0.003669962],"genre_scores_gemma":[0.8472446,0.001182345,0.1311912,0.0003429922,0.0001592239,0.0006642083,0.0173384,0.0003109545,0.001566161],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01071046,"threshold_uncertainty_score":0.04164201,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09860381484964052,"score_gpt":0.4511032186632735,"score_spread":0.352499403813633,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}