{"id":"W4401794356","doi":"10.1016/j.jclinane.2024.111582","title":"The evaluation of the performance of ChatGPT in the management of labor analgesia","year":2024,"lang":"en","type":"article","venue":"Journal of Clinical Anesthesia","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":false,"ca_institutions":"Mount Sinai Hospital","funders":"","keywords":"Medicine; Chatbot; Anesthesia; Intensive care medicine; Natural language processing; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01026348,0.00005253885,0.0002417921,0.00005280905,0.00003487643,0.000006929227,0.0002329145,0.00005741487,0.00002194969],"category_scores_gemma":[0.0003003189,0.00002146821,0.0001830859,0.0003982005,0.0001535273,0.00005452808,0.000009308187,0.0002965227,0.000002367502],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002407937,"about_ca_system_score_gemma":0.000368327,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002317151,"about_ca_topic_score_gemma":0.000008502486,"domain_scores_codex":[0.9972637,0.0003740698,0.001463005,0.00006456903,0.0007487911,0.00008580789],"domain_scores_gemma":[0.9978335,0.0008000714,0.000576598,0.000238208,0.0005239709,0.00002767467],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0006933881,0.0005280635,0.3106419,0.0008641369,0.0001860104,0.00002244338,0.003332677,0.0003453761,0.0000889574,0.003329825,0.0009291699,0.679038],"study_design_scores_gemma":[0.0001180724,0.0007252723,0.9821177,0.001447212,0.0003593047,0.00009182202,0.002896936,0.004879009,0.002105608,0.001541792,0.003687365,0.00002988539],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.973871,0.002054464,0.000004160044,0.02337808,0.0002790842,0.000252625,2.006214e-7,0.000001094924,0.0001592777],"genre_scores_gemma":[0.9972434,0.002172106,0.00006926661,0.000227514,0.0002170444,0.000004104794,2.528998e-7,0.000004254513,0.00006204134],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6790082,"threshold_uncertainty_score":0.3557138,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2566988728586553,"score_gpt":0.5249742138267146,"score_spread":0.2682753409680593,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}