{"id":"W4416224769","doi":"10.1038/s44277-025-00049-6","title":"Mindbench.ai: an actionable platform to evaluate the profile and performance of large language models in a mental healthcare context","year":2025,"lang":"en","type":"article","venue":"NPP—Digital Psychiatry and Neuroscience","topic":"Digital Mental Health Interventions","field":"Psychology","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre for Addiction and Mental Health","funders":"National Institutes of Health; National Institute for Health and Care Research; Department of Health and Social Care; Wellcome Trust; Argosy Foundation","keywords":"General partnership; Context (archaeology); Mental health; Benchmarking; Health care; Scale (ratio); Resource (disambiguation); Empirical research","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002757267,0.0001353468,0.0001534579,0.0001495543,0.0002136935,0.0001016293,0.0002288813,0.00004957112,0.00001814534],"category_scores_gemma":[0.0000180103,0.0001079862,0.00003246089,0.0004570835,0.000162273,0.001100802,0.0001264528,0.0001808944,0.000008674408],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003028877,"about_ca_system_score_gemma":0.00008864832,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000991793,"about_ca_topic_score_gemma":0.0003202144,"domain_scores_codex":[0.9986538,0.00003738591,0.0003397911,0.00042289,0.0001877505,0.0003583822],"domain_scores_gemma":[0.999488,0.0000321314,0.00005985846,0.0002603178,0.00002910503,0.0001306137],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.003303177,0.006079105,0.2277172,0.001522741,0.00003618541,0.00001657262,0.01812007,0.00006144909,0.001803446,0.4197793,0.006160115,0.3154006],"study_design_scores_gemma":[0.007086948,0.01494963,0.8760735,0.002039623,0.00004116538,0.0002886823,0.03754868,0.02527299,0.003014931,0.0259127,0.006582666,0.001188443],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.985886,0.0004784352,0.00003531035,0.003696975,0.001337707,0.0006627647,0.0002774245,0.00002114506,0.007604282],"genre_scores_gemma":[0.9949148,0.00000747188,0.00002772989,0.003261096,0.0000156878,0.00007882238,0.000008496178,0.000007808323,0.00167812],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6483563,"threshold_uncertainty_score":0.440355,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03955984231817397,"score_gpt":0.3933695669102549,"score_spread":0.3538097245920809,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}