{"id":"W4402181615","doi":"10.1016/j.jhsg.2024.07.011","title":"Large Language Models in the Diagnosis of Hand and Peripheral Nerve Injuries: An Evaluation of ChatGPT and the Isabel Differential Diagnosis Generator","year":2024,"lang":"en","type":"article","venue":"Journal of Hand Surgery Global Online","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; University of Toronto","funders":"","keywords":"Generator (circuit theory); Peripheral nerve; Differential diagnosis; Peripheral; Differential (mechanical device); Medicine; Computer science; Physical medicine and rehabilitation; Medical emergency; Anatomy; Engineering; Pathology; Physics; Internal medicine; Aerospace engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008893339,0.001216029,0.0008015717,0.001919729,0.0004660464,0.001595238,0.001637464,0.001168456,0.002521014],"category_scores_gemma":[0.03305876,0.0003500234,0.0007983868,0.0007937995,0.000516072,0.001807559,0.001972181,0.001443408,0.0006947443],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001493299,"about_ca_system_score_gemma":0.001574396,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008770163,"about_ca_topic_score_gemma":0.008527797,"domain_scores_codex":[0.9966217,0.002062409,0.0002633924,0.0005185814,0.0004262798,0.0001076368],"domain_scores_gemma":[0.9512402,0.04438937,0.0009658113,0.001054565,0.001687605,0.0006625001],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01076833,0.002174296,0.114216,0.001346515,0.00112051,0.001264868,0.001614042,0.1781875,0.006904771,0.001496667,0.008838406,0.6720681],"study_design_scores_gemma":[0.0004361801,0.001361659,0.01263861,0.00008652834,0.0002890625,0.0006537678,0.0005275926,0.9767933,0.004311752,0.001502229,0.001347081,0.00005226801],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9099069,0.001150996,0.07400675,0.00126661,0.0001840258,0.0008165002,0.001526941,0.007457841,0.003683437],"genre_scores_gemma":[0.8854622,0.0002656506,0.1096072,0.0004526586,0.00006805122,0.0003703244,0.002495788,0.0001597047,0.001118459],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008893339,"threshold_uncertainty_score":0.04703307,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1126998629274872,"score_gpt":0.4197429704993473,"score_spread":0.3070431075718602,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}