{"id":"W4401794356","doi":"10.1016/j.jclinane.2024.111582","title":"The evaluation of the performance of ChatGPT in the management of labor analgesia","year":2024,"lang":"en","type":"article","venue":"Journal of Clinical Anesthesia","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"Mount Sinai Hospital","funders":"","keywords":"Medicine; Chatbot; Anesthesia; Intensive care medicine; Natural language processing; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002515529,0.000321439,0.0004839048,0.0004701985,0.0002759218,0.0006079394,0.0004155455,0.0006055021,0.001332431],"category_scores_gemma":[0.01691793,0.00007847635,0.0002103234,0.0003734108,0.0004670746,0.0004166718,0.0003831089,0.0004721599,0.0003392932],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004058789,"about_ca_system_score_gemma":0.0009382889,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002274097,"about_ca_topic_score_gemma":0.002228429,"domain_scores_codex":[0.9983245,0.0009154514,0.0001329969,0.0001127955,0.0003974524,0.0001167065],"domain_scores_gemma":[0.9899073,0.006370284,0.0008085772,0.0002440731,0.0015908,0.001078915],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.07107245,0.006550785,0.401821,0.001689065,0.0005883394,0.001837939,0.004803931,0.006254007,0.06110037,0.0006227071,0.001775276,0.4418842],"study_design_scores_gemma":[0.0009392786,0.1012978,0.8283772,0.0002260377,0.001026438,0.002368953,0.002997459,0.02319672,0.0345719,0.0002991592,0.004599473,0.00009958325],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9971511,0.0004494368,0.0006843172,0.0001069808,0.00003359684,0.00006823247,0.00008014992,0.00003040257,0.001395788],"genre_scores_gemma":[0.9977371,0.0002166164,0.001255391,0.00005861869,0.00003695399,0.00004743376,0.00008540332,0.00000915274,0.0005532595],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002515529,"threshold_uncertainty_score":0.01330352,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2566988728586553,"score_gpt":0.5249742138267146,"score_spread":0.2682753409680593,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}