{"id":"W4409451352","doi":"10.5435/jaaos-d-24-01049","title":"Comparison of ChatGPT's Diagnostic and Management Accuracy of Foot and Ankle Bone–Related Pathologies to Orthopaedic Surgeons","year":2025,"lang":"en","type":"article","venue":"Journal of the American Academy of Orthopaedic Surgeons","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Object Research Systems (Canada)","funders":"","keywords":"Medicine; Foot (prosody); Orthopedic surgery; Ankle; Foot and ankle surgery; Physical therapy; Sports medicine; Surgery","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005935673,0.0003787031,0.0003938039,0.003092166,0.0004195059,0.001017258,0.001038877,0.0006228043,0.002831636],"category_scores_gemma":[0.05696895,0.0002961473,0.0006638523,0.001345723,0.0008492448,0.001009601,0.002234434,0.0004415889,0.0007346176],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008582821,"about_ca_system_score_gemma":0.0006786117,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003658219,"about_ca_topic_score_gemma":0.004733452,"domain_scores_codex":[0.994984,0.001862563,0.0007521653,0.0005652676,0.001452397,0.000383611],"domain_scores_gemma":[0.9581712,0.02113429,0.009305455,0.001719461,0.007096768,0.002572864],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003456733,0.00006354105,0.9875692,0.00006857242,0.00007025234,0.0001348951,0.0006374723,0.0002673574,0.0002493992,0.00003537658,0.0003925121,0.01016577],"study_design_scores_gemma":[0.00002232527,0.000343332,0.9932914,0.00006348751,0.00006147854,0.000846266,0.001331279,0.002913581,0.0004164204,0.0000936034,0.000597268,0.00001952388],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9973725,0.0001868378,0.0002932668,0.00008577501,0.00003403831,0.00003810184,0.0002670943,0.00002752438,0.001694863],"genre_scores_gemma":[0.9988403,0.00009845575,0.0004101014,0.00002017363,0.00002110643,0.00002354094,0.0003550994,0.000006094957,0.0002252652],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005935673,"threshold_uncertainty_score":0.0313912,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06551089101692614,"score_gpt":0.4089501639106905,"score_spread":0.3434392728937644,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}