{"id":"W4394873790","doi":"10.1007/s00464-024-10807-w","title":"The performance of artificial intelligence large language model-linked chatbots in surgical decision-making for gastroesophageal reflux disease","year":2024,"lang":"en","type":"article","venue":"Surgical Endoscopy","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":24,"is_retracted":false,"has_abstract":false,"ca_institutions":"McMaster University","funders":"","keywords":"Perplexity; GERD; Guideline; Medicine; General surgery; Reflux; Medical emergency; Disease; Internal medicine; Language model; Computer science; Pathology; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001100026,0.0001749361,0.0003104127,0.0001570972,0.0001721247,0.00004993624,0.0001677274,0.0001123479,0.0001237892],"category_scores_gemma":[0.0004596245,0.0001205705,0.0001878963,0.0004375381,0.0001399499,0.0001229711,0.00004861473,0.0003636407,0.00004251188],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001186652,"about_ca_system_score_gemma":0.0004231774,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002260022,"about_ca_topic_score_gemma":0.00003887218,"domain_scores_codex":[0.997799,0.00006199782,0.0008151021,0.0003864319,0.0003859554,0.000551499],"domain_scores_gemma":[0.997292,0.001971164,0.00007115811,0.0003264499,0.0001257626,0.0002134106],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01582315,0.0008077315,0.007812962,0.001219716,0.00005665565,0.001110232,0.004944522,0.004907233,0.0002664482,0.04381223,0.0002044824,0.9190347],"study_design_scores_gemma":[0.0002275958,0.0004263988,0.0003994602,0.001377637,0.00004938111,0.00003957526,0.001020424,0.9794486,0.004584211,0.01001635,0.00223839,0.0001719726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9890688,0.003345488,0.005179382,0.0008023111,0.0006938004,0.000657453,0.00003907976,0.00006314607,0.000150522],"genre_scores_gemma":[0.9972481,0.0005896131,0.00123275,0.00005472016,0.0005418865,0.0001338859,0.00003150223,0.0000278383,0.0001397371],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9745414,"threshold_uncertainty_score":0.4916722,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06467560335054845,"score_gpt":0.4351248248935671,"score_spread":0.3704492215430187,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}