{"id":"W4416185868","doi":"10.1038/s44277-025-00042-z","title":"Using large language models as a scalable mental status evaluation technique","year":2025,"lang":"en","type":"article","venue":"NPP—Digital Psychiatry and Neuroscience","topic":"Mental Health via Writing","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Mental health; Scalability; Language model; Identification (biology); Anxiety; Natural language; Random forest; Computational linguistics; Session (web analytics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005696929,0.000171966,0.0001512157,0.0001743662,0.0004293459,0.0001733848,0.0001908106,0.00009129185,0.00008870653],"category_scores_gemma":[0.00005636365,0.0001763372,0.00004086775,0.0005810926,0.0001351986,0.0007392699,0.0001370563,0.0002181348,0.00002197587],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000816226,"about_ca_system_score_gemma":0.0002346682,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008677533,"about_ca_topic_score_gemma":0.0000122461,"domain_scores_codex":[0.9979965,0.0001055063,0.0003081808,0.000626145,0.0003511642,0.0006124717],"domain_scores_gemma":[0.9993671,0.00003757236,0.0001001703,0.0002998165,0.00003466661,0.0001606979],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001273952,0.005675603,0.2085722,0.001166458,0.000056948,0.0001062529,0.009278535,0.0004246095,0.1880616,0.4720153,0.009328354,0.1040403],"study_design_scores_gemma":[0.02364894,0.004984207,0.1018857,0.003711903,0.000398439,0.00269258,0.0579036,0.4909248,0.02131846,0.2573265,0.02990722,0.005297579],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8624242,0.001005344,0.004557931,0.0003112518,0.002309598,0.001034866,0.0001040745,0.0001105162,0.1281422],"genre_scores_gemma":[0.9957101,0.00001217679,0.0007836029,0.002215326,0.0000491542,0.00007055559,0.00001032248,0.00001483338,0.001133931],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4905002,"threshold_uncertainty_score":0.7190821,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05727910150430604,"score_gpt":0.4208321513042883,"score_spread":0.3635530497999823,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}