{"id":"W4416224769","doi":"10.1038/s44277-025-00049-6","title":"Mindbench.ai: an actionable platform to evaluate the profile and performance of large language models in a mental healthcare context","year":2025,"lang":"en","type":"article","venue":"NPP—Digital Psychiatry and Neuroscience","topic":"Digital Mental Health Interventions","field":"Psychology","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre for Addiction and Mental Health","funders":"National Institutes of Health; National Institute for Health and Care Research; Department of Health and Social Care; Wellcome Trust; Argosy Foundation","keywords":"General partnership; Context (archaeology); Mental health; Benchmarking; Health care; Scale (ratio); Resource (disambiguation); Empirical research","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02477794,0.002599856,0.0006977624,0.006762893,0.001004141,0.005918221,0.003139026,0.002319732,0.01138345],"category_scores_gemma":[0.08688954,0.0007670263,0.001307446,0.001929255,0.001301261,0.009933534,0.009949041,0.00288476,0.00452395],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001726646,"about_ca_system_score_gemma":0.003177095,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003192248,"about_ca_topic_score_gemma":0.004628855,"domain_scores_codex":[0.983164,0.009443878,0.001572236,0.001403843,0.003803749,0.0006122146],"domain_scores_gemma":[0.9311978,0.04750409,0.004125542,0.007614008,0.007006672,0.002551873],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004146899,0.004206177,0.04449033,0.004389136,0.0007088109,0.001134538,0.01378753,0.02294274,0.02230579,0.05105234,0.2431416,0.5876942],"study_design_scores_gemma":[0.001631327,0.004759458,0.04019014,0.002242361,0.0004197596,0.0009638318,0.00727393,0.3543979,0.04084366,0.1416339,0.4044373,0.001206435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2010034,0.001533889,0.4698148,0.007200543,0.001200899,0.009770599,0.04610106,0.1935309,0.06984372],"genre_scores_gemma":[0.3204952,0.0006583062,0.5929109,0.002101846,0.0002589818,0.008645687,0.05341987,0.009976995,0.0115321],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02477794,"threshold_uncertainty_score":0.1310398,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03955984231817397,"score_gpt":0.3933695669102549,"score_spread":0.3538097245920809,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}