{"id":"W4411902008","doi":"10.1186/s12883-025-04280-8","title":"Evaluating ChatGPT and DeepSeek in postdural puncture headache management: a comparative study with international consensus guidelines","year":2025,"lang":"en","type":"article","venue":"BMC Neurology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hospital for Sick Children","funders":"","keywords":"Medicine; Neurosurgery; Neurology; MEDLINE; Neurochemistry; Surgery; Psychiatry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003167932,0.0001006934,0.0001993354,0.0001925429,0.00005388361,0.00001129148,0.00005705585,0.00004884622,0.00002120641],"category_scores_gemma":[0.0001911501,0.00007737998,0.00001428673,0.0001879262,0.00007311562,0.00002210296,0.00004150046,0.0001881845,0.000005001042],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002293936,"about_ca_system_score_gemma":0.0001098879,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000327352,"about_ca_topic_score_gemma":0.002735974,"domain_scores_codex":[0.9989541,0.000151001,0.0003397832,0.0002789589,0.0001252422,0.000150886],"domain_scores_gemma":[0.999306,0.0002465161,0.00006115209,0.0001336417,0.0002163527,0.00003637089],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002680313,0.0002712224,0.9809135,0.00006387083,0.00006190366,0.00007706925,0.002504589,0.0005680203,0.00008610282,0.0005168937,0.0005653256,0.01169122],"study_design_scores_gemma":[0.001049638,0.00305209,0.9555434,0.0001077555,0.0001208095,0.0001724923,0.01253036,0.02456582,0.0002041078,0.00157814,0.0009406552,0.0001347056],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9796327,0.0001658414,0.0001420364,0.01796204,0.0003292655,0.0008555893,9.425999e-7,0.00002005353,0.0008915056],"genre_scores_gemma":[0.9943191,0.00001119979,0.001618407,0.003620408,0.00007969086,0.00007129074,0.00000858167,0.000005114564,0.0002661995],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02537005,"threshold_uncertainty_score":0.3155464,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3626240455689329,"score_gpt":0.5386386789387114,"score_spread":0.1760146333697785,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}