{"id":"W4409341463","doi":"10.2196/66126","title":"Feasibility of a Randomized Controlled Trial of Large AI-Based Linguistic Models for Clinical Reasoning Training of Physical Therapy Students: Pilot Randomized Parallel-Group Study","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Training (meteorology); Randomized controlled trial; Effi; Artificial intelligence; Computer science; Psychology; Physical therapy; Natural language processing; Medicine; World Wide Web; Surgery","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03092424,0.00375124,0.007993672,0.001962869,0.002290706,0.002792064,0.003357737,0.005939259,0.01638371],"category_scores_gemma":[0.02978669,0.002551843,0.003219629,0.001680289,0.004687067,0.004189996,0.001964964,0.004749187,0.002776508],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002755622,"about_ca_system_score_gemma":0.006490087,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001501057,"about_ca_topic_score_gemma":0.001896975,"domain_scores_codex":[0.9703274,0.02096499,0.002144017,0.002423309,0.00212672,0.002013553],"domain_scores_gemma":[0.9735913,0.01348615,0.003695089,0.003012242,0.002986549,0.003228585],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"randomized_trial","study_design_gemma":"randomized_trial","study_design_scores_codex":[0.9418289,0.04984305,0.0003640938,0.000631907,0.0003888671,0.00003853097,0.0001448075,0.0002952646,0.0008065408,0.0001688576,0.0002891852,0.005199923],"study_design_scores_gemma":[0.8494191,0.1484626,0.0005252573,0.00004221205,0.0002097579,0.000006665324,0.00005698363,0.000454173,0.0002502156,0.0001988922,0.0003580844,0.00001606785],"study_design_candidate":"randomized_trial","study_design_consensus":"randomized_trial","genre_codex":"protocol","genre_gemma":"empirical","genre_scores_codex":[0.4437444,0.0009361911,0.005259709,0.001003641,0.00256699,0.5426505,0.001037479,0.0003873701,0.002413776],"genre_scores_gemma":[0.4030342,0.0005180788,0.01667669,0.001259804,0.0007960267,0.5750084,0.0003808466,0.00004696954,0.002278917],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03092424,"threshold_uncertainty_score":0.163545,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1683317366160802,"score_gpt":0.5576376589262059,"score_spread":0.3893059223101257,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}