{"id":"W4400904138","doi":"10.2196/57306","title":"Exploring the Efficacy of Large Language Models in Summarizing Mental Health Counseling Sessions: Benchmark Study","year":2024,"lang":"en","type":"article","venue":"JMIR Mental Health","topic":"Mental Health via Writing","field":"Psychology","cited_by":29,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Computer science; Benchmarking; Mental health; Benchmark (surveying); Task (project management); Recall; Process (computing); Applied psychology; Psychology; Medical education; Artificial intelligence; Medicine; Psychotherapist; Cognitive psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.004148182,0.0003334031,0.0005949665,0.0003589869,0.000694981,0.00005205647,0.0003627938,0.00006185585,0.0002177363],"category_scores_gemma":[0.0000119573,0.0002748218,0.000115255,0.0008179932,0.00005711103,0.0004022319,0.0002046325,0.0008199122,0.00006428201],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001232241,"about_ca_system_score_gemma":0.0003761683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006395079,"about_ca_topic_score_gemma":0.001995515,"domain_scores_codex":[0.9947661,0.00110038,0.001451025,0.0007459539,0.0006751443,0.00126145],"domain_scores_gemma":[0.9983745,0.0004372105,0.0003008965,0.0005547966,0.00001696342,0.0003156575],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0004942505,0.004254329,0.00876333,0.001941542,0.0001074382,0.00007910806,0.859122,0.00003540728,0.0001611759,0.005341313,0.006515165,0.113185],"study_design_scores_gemma":[0.006927067,0.002252304,0.04836503,0.006765486,0.0000139785,0.00008886388,0.9244745,0.005582522,0.00005875746,0.000144986,0.004693412,0.0006330807],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9764007,0.01157973,0.00003099091,0.003590794,0.002313096,0.004619222,0.0002473138,0.0001739358,0.001044254],"genre_scores_gemma":[0.9962739,0.0005107074,0.0001821693,0.001675519,0.0002193635,0.0006658892,0.0001679418,0.00008022749,0.0002243354],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1125519,"threshold_uncertainty_score":0.9999704,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09860381484964052,"score_gpt":0.4511032186632735,"score_spread":0.352499403813633,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}