{"id":"W7106794673","doi":"10.2196/80167","title":"Digitally Assisted Clinical Decision-Making in Traditional Chinese Medicine: Comparative Study of 5 Large Language Models","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Traditional Chinese Medicine Studies","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Government of Jiangsu Province; National Natural Science Foundation of China","keywords":"Standardization; Reliability (semiconductor); Quality (philosophy); Benchmark (surveying); Benchmarking; Medical prescription; Clinical trial; MEDLINE","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01434015,0.0006940174,0.0006608205,0.002230914,0.000601134,0.003671879,0.001168584,0.0008501005,0.002433335],"category_scores_gemma":[0.04873436,0.0002993291,0.001334342,0.001487307,0.0007912461,0.002945602,0.001982043,0.0009047881,0.0003123641],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003826921,"about_ca_system_score_gemma":0.002582595,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009404399,"about_ca_topic_score_gemma":0.006714896,"domain_scores_codex":[0.99496,0.003489975,0.0003596531,0.0004884327,0.0005442397,0.0001576804],"domain_scores_gemma":[0.9159312,0.07712631,0.002138847,0.001711127,0.002098189,0.0009943242],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.007311436,0.003831748,0.1386345,0.001868761,0.001545938,0.0008215958,0.009450492,0.3223567,0.003073192,0.01696259,0.003972877,0.4901702],"study_design_scores_gemma":[0.0003398567,0.001694755,0.03167447,0.0002744485,0.0005779675,0.0001823116,0.002894787,0.9454238,0.001894163,0.01166572,0.003234425,0.0001433846],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9749254,0.0007141305,0.0181899,0.0006297016,0.00003134653,0.0003659234,0.0003232605,0.0002748685,0.004545504],"genre_scores_gemma":[0.9814969,0.0003349222,0.01709537,0.00009732811,0.00001318794,0.0001894483,0.0003605297,0.00002820883,0.0003839914],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01434015,"threshold_uncertainty_score":0.07583886,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1798021339684095,"score_gpt":0.5470641326485618,"score_spread":0.3672619986801523,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}