{"id":"W2982702361","doi":"10.2196/13863","title":"Accuracy of a Chatbot (Ada) in the Diagnosis of Mental Disorders: Comparative Case Study With Lay and Expert Users","year":2019,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Digital Mental Health Interventions","field":"Psychology","cited_by":136,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Kappa; Mental health; Medical diagnosis; Inter-rater reliability; Diagnostic Classification of Mental Health and Developmental Disorders of Infancy and Early Childhood; Psychology; Psychiatry; Medicine; Clinical psychology; Prevalence of mental disorders; Developmental psychology; Rating scale; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02196644,0.0005263035,0.000669045,0.004220391,0.002480773,0.001797587,0.001768278,0.00227655,0.001785174],"category_scores_gemma":[0.113489,0.0009081872,0.0008863757,0.001040299,0.002914575,0.002074415,0.003314139,0.001327831,0.0004757296],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002055114,"about_ca_system_score_gemma":0.0009387289,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007844773,"about_ca_topic_score_gemma":0.01033129,"domain_scores_codex":[0.9778104,0.01363912,0.001949568,0.001810411,0.003665385,0.001125124],"domain_scores_gemma":[0.8541873,0.1161226,0.008973155,0.00540606,0.01297434,0.002336649],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007961389,0.001639575,0.7651808,0.0005140931,0.0001114119,0.01471317,0.1906746,0.0002540133,0.001582806,0.0002079722,0.0003885876,0.02393675],"study_design_scores_gemma":[0.0001117918,0.005136876,0.7468967,0.0005718264,0.0002963226,0.04914977,0.1846525,0.006667694,0.003742207,0.0004434804,0.002128142,0.0002026955],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9989455,0.0001184685,0.0002635578,0.0000678065,0.000005602573,0.00006956637,0.00002262398,0.000004224816,0.0005027036],"genre_scores_gemma":[0.9988826,0.0001206103,0.0006911904,0.00004524271,0.00001088445,0.00004748364,0.00002881894,0.000005226033,0.0001680331],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02196644,"threshold_uncertainty_score":0.1161711,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1705206091532488,"score_gpt":0.5455406751274376,"score_spread":0.3750200659741889,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}