{"id":"W4390776874","doi":"10.2196/51308","title":"Comprehensiveness, Accuracy, and Readability of Exercise Recommendations Provided by an AI-Based Chatbot: Mixed Methods Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Mobile Health and mHealth Applications","field":"Health Professions","cited_by":54,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"CVS Health; University of Connecticut","keywords":"Readability; Chatbot; Medicine; Health care; Medical education; Physical therapy; Computer science; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003337048,0.0001875132,0.000413478,0.0001794814,0.0005396522,0.00001796656,0.0002228495,0.000298734,0.001041543],"category_scores_gemma":[0.001692101,0.0001642477,0.00004274214,0.0005493849,0.0001263114,0.0002340321,0.00007022559,0.0008220606,0.00003133585],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001839677,"about_ca_system_score_gemma":0.008348808,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001104336,"about_ca_topic_score_gemma":0.0002264344,"domain_scores_codex":[0.9952848,0.00216188,0.001150293,0.0005902356,0.0004142076,0.0003985877],"domain_scores_gemma":[0.9952993,0.002416755,0.0002760162,0.0006490482,0.0004442832,0.0009145949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008146802,0.002596342,0.008713541,0.003239494,0.00001096505,2.602218e-7,0.003521312,3.005499e-7,0.0001312601,0.001117949,0.1059252,0.8746619],"study_design_scores_gemma":[0.00128739,0.0006320638,0.07563359,0.00267882,0.000154909,0.00000286801,0.02088446,0.00867191,0.0002580537,0.004071902,0.8853049,0.00041911],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7860647,0.005118252,0.02023681,0.1488335,0.005693199,0.03256866,0.000219943,0.0007303619,0.0005345186],"genre_scores_gemma":[0.9389603,0.0004995443,0.007513538,0.004933603,0.0003074589,0.04647301,0.0009481491,0.00005019644,0.0003141664],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8742428,"threshold_uncertainty_score":0.9998716,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05046909020929952,"score_gpt":0.5520507845820172,"score_spread":0.5015816943727176,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}