{"id":"W4416702500","doi":"10.2196/87887","title":"From Evaluation to Enhancement: A Decision Support Framework for Quality Assurance in Therapeutic AI Systems (Preprint)","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Digital Mental Health Interventions","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Quality assurance; Chatbot; Quality (philosophy); Clinical decision support system; Decision support system; Baseline (sea); Quality management","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1620164,0.001718781,0.001236215,0.005249242,0.003364688,0.01473784,0.004259559,0.00378796,0.007198142],"category_scores_gemma":[0.1635625,0.001135053,0.002621258,0.00258943,0.01160517,0.01306553,0.01046553,0.00379681,0.001216946],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008713312,"about_ca_system_score_gemma":0.01770968,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005067958,"about_ca_topic_score_gemma":0.004523456,"domain_scores_codex":[0.8575408,0.1161906,0.009882968,0.005237731,0.009402331,0.001745412],"domain_scores_gemma":[0.8060842,0.153927,0.009445387,0.01084878,0.01485057,0.004844026],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0007211139,0.0009807849,0.006676394,0.001995746,0.0002907454,0.0006280883,0.01811367,0.03847369,0.003521044,0.5033198,0.01757005,0.4077089],"study_design_scores_gemma":[0.0006218828,0.0008964768,0.002053651,0.002629628,0.0002270635,0.0002746492,0.00753889,0.3142575,0.007618515,0.5948545,0.06877209,0.0002552288],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007255884,0.0002545686,0.9717229,0.009723258,0.0001143871,0.002474663,0.0002253616,0.001718911,0.006510035],"genre_scores_gemma":[0.08070643,0.0001092498,0.9160415,0.0006307603,0.00004461211,0.001563412,0.0002196939,0.0001197106,0.0005645155],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8379836,"threshold_uncertainty_score":0.856835,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0700789358171249,"score_gpt":0.5153249240570671,"score_spread":0.4452459882399422,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}