{"id":"W7108214372","doi":"10.48620/92750","title":"How Well Does ChatGPT-4o Reason? Expert Evaluation of Diagnostic and Therapeutic Performance in Hand Surgery.","year":2025,"lang":"en","type":"article","venue":"Open Access CRIS of the University of Bern","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hand and Upper Limb Clinic","funders":"","keywords":"Carpal tunnel syndrome; Readability; English language; Relevance (law); Diagnostic test; MEDLINE; Clinical decision making; Clinical Practice","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02123916,0.000519921,0.0004172595,0.001174608,0.0005316336,0.001200299,0.0008263214,0.0009671588,0.002133313],"category_scores_gemma":[0.09942217,0.0002117293,0.0005259133,0.0003595097,0.001027936,0.001079055,0.002260287,0.0006247054,0.0005305727],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001374831,"about_ca_system_score_gemma":0.001602238,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00105137,"about_ca_topic_score_gemma":0.002236403,"domain_scores_codex":[0.9851002,0.01065361,0.001145452,0.000926388,0.001662845,0.000511512],"domain_scores_gemma":[0.9075804,0.06737698,0.0110858,0.00339049,0.007228357,0.003338029],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002647412,0.001396715,0.799198,0.0009560317,0.0002013852,0.001856473,0.04684107,0.006413267,0.00952038,0.0006334039,0.003863148,0.1264728],"study_design_scores_gemma":[0.0006703902,0.01043557,0.7823125,0.0009504483,0.0003703954,0.009762975,0.07198071,0.08566833,0.01753042,0.005602142,0.01418998,0.0005261772],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9942029,0.00007237229,0.003285245,0.0002644591,0.00001757265,0.0002638077,0.0001051327,0.000070516,0.001718033],"genre_scores_gemma":[0.99216,0.00004610224,0.007055046,0.0001144339,0.00001045429,0.0001941188,0.0001371443,0.00001149681,0.0002711598],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02123916,"threshold_uncertainty_score":0.1123248,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1582755051593802,"score_gpt":0.4228446688431823,"score_spread":0.2645691636838021,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}