{"id":"W4381486753","doi":"10.1016/j.hansur.2023.06.005","title":"Is ChatGPT able to pass the first part of the European Board of Hand Surgery diploma examination?","year":2023,"lang":"en","type":"letter","venue":"Hand surgery & rehabilitation","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":17,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University","funders":"","keywords":"CLARITY; Turkish; Medicine; Quality (philosophy); Conversation; Medical education; Medical physics; Psychology; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.002232739,0.0002713954,0.0007012566,0.0004937085,0.0050034,0.002305245,0.001087416,0.02447421,0.03054129],"category_scores_gemma":[0.01374679,0.0003835807,0.0005197391,0.0003828871,0.001596677,0.002493637,0.001809398,0.01321089,0.01030879],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003436577,"about_ca_system_score_gemma":0.01024353,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03078434,"about_ca_topic_score_gemma":0.06334087,"domain_scores_codex":[0.9981728,0.0002332063,0.0001305158,0.0001855501,0.0003991453,0.0008787878],"domain_scores_gemma":[0.9938444,0.001611119,0.0002996891,0.0001177052,0.000945531,0.003181525],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.0001064846,0.00007423163,0.007618544,0.00004309427,0.000008766594,0.007574483,0.0006657626,0.0000452419,0.0004281451,0.003384793,0.9637507,0.0162998],"study_design_scores_gemma":[0.0001773603,0.0003117818,0.017884,0.0006277377,0.00002696692,0.01668569,0.008818365,0.000632436,0.0003961886,0.007245354,0.9470872,0.0001068923],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.004350239,0.0003613634,0.00008645321,0.9701685,0.006884722,0.00003035465,0.0001319538,0.00004022412,0.01794609],"genre_scores_gemma":[0.04073875,0.0009986538,0.0003180561,0.8760077,0.00613333,0.0001113471,0.0001555758,0.00006380796,0.0754728],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9977673,"threshold_uncertainty_score":0.1021708,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1020060519604576,"score_gpt":0.3404105089444738,"score_spread":0.2384044569840162,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}