{"id":"W4411902008","doi":"10.1186/s12883-025-04280-8","title":"Evaluating ChatGPT and DeepSeek in postdural puncture headache management: a comparative study with international consensus guidelines","year":2025,"lang":"en","type":"article","venue":"BMC Neurology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hospital for Sick Children","funders":"","keywords":"Medicine; Neurosurgery; Neurology; MEDLINE; Neurochemistry; Surgery; Psychiatry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1402163,0.0005420609,0.003134381,0.006494789,0.001188797,0.003396897,0.0020038,0.002313615,0.001642498],"category_scores_gemma":[0.381473,0.0006724127,0.00400761,0.008777063,0.001786393,0.003793014,0.004128338,0.002355621,0.0003283876],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006485762,"about_ca_system_score_gemma":0.01307601,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006468829,"about_ca_topic_score_gemma":0.008395045,"domain_scores_codex":[0.8140005,0.1128758,0.03966321,0.004868744,0.02641395,0.002177623],"domain_scores_gemma":[0.5408207,0.2666875,0.07168824,0.01080307,0.1029065,0.007093995],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00907015,0.003226342,0.3292205,0.06048042,0.008635662,0.0004409293,0.06196554,0.002230298,0.00058086,0.003215454,0.01437375,0.5065601],"study_design_scores_gemma":[0.01173404,0.01980354,0.6922245,0.0824832,0.01676816,0.001436053,0.09046273,0.01116032,0.001661691,0.004867134,0.06621546,0.001183159],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8758625,0.05587716,0.01160329,0.007286376,0.001308872,0.02099685,0.00334134,0.0002496628,0.02347391],"genre_scores_gemma":[0.9462521,0.01091865,0.02582293,0.001889579,0.0002318475,0.01279975,0.001573178,0.00006081141,0.0004512297],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1402163,"threshold_uncertainty_score":0.7415435,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3626240455689329,"score_gpt":0.5386386789387114,"score_spread":0.1760146333697785,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}