{"id":"W7108214372","doi":"10.48620/92750","title":"How Well Does ChatGPT-4o Reason? Expert Evaluation of Diagnostic and Therapeutic Performance in Hand Surgery.","year":2025,"lang":"en","type":"article","venue":"Open Access CRIS of the University of Bern","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hand and Upper Limb Clinic","funders":"","keywords":"Carpal tunnel syndrome; Readability; English language; Relevance (law); Diagnostic test; MEDLINE; Clinical decision making; Clinical Practice","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005921213,0.0000516794,0.0002014886,0.0001109551,0.00008654912,0.00002549791,0.0003254034,0.00004535236,0.00007628584],"category_scores_gemma":[0.0003638834,0.00003875132,0.00003498896,0.0002260089,0.0001688953,0.0003098529,0.0001912339,0.00006004233,5.138716e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005387245,"about_ca_system_score_gemma":0.0002418728,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009786471,"about_ca_topic_score_gemma":0.001638119,"domain_scores_codex":[0.9993922,0.00009792644,0.0001424089,0.0001145086,0.0001785324,0.00007442903],"domain_scores_gemma":[0.9989005,0.0004167035,0.0001570704,0.0002194517,0.000284957,0.0000213642],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001902584,0.0001219969,0.8477975,0.0002303261,0.00002842686,3.159666e-7,0.002602656,0.00002962432,0.0004372922,0.00004246737,0.0003929301,0.1481262],"study_design_scores_gemma":[0.0002210811,0.00004615817,0.882648,0.001219127,0.0001584737,6.665172e-7,0.009044432,0.002025909,0.1023199,0.000647103,0.001603801,0.00006541426],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9905283,0.0006996699,0.00002650259,0.007185075,0.0001518612,0.0004918053,0.000002222832,0.00000158196,0.0009129283],"genre_scores_gemma":[0.9978124,0.001406361,0.00002441329,0.00006340574,0.000009020428,0.000001141177,0.000002636606,0.000002488438,0.0006781197],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1480608,"threshold_uncertainty_score":0.9968075,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1582755051593802,"score_gpt":0.4228446688431823,"score_spread":0.2645691636838021,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}