{"id":"W4407237982","doi":"10.1016/j.bja.2025.01.001","title":"A comparison of large language model-generated and published perioperative neurocognitive disorder recommendations: a cross-sectional web-based analysis","year":2025,"lang":"en","type":"article","venue":"British Journal of Anaesthesia","topic":"Intensive Care Unit Cognitive Disorders","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute of General Medical Sciences; National Institutes of Health","keywords":"Neurocognitive; Cross-sectional study; Psychology; Medicine; Clinical psychology; Psychiatry; Cognition; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07387371,0.0005918465,0.001242194,0.00854928,0.000420488,0.002695312,0.001204839,0.0009626875,0.001829024],"category_scores_gemma":[0.2796347,0.0005365479,0.002876415,0.007854045,0.0006257634,0.002454367,0.00182899,0.001106376,0.0003811575],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001951936,"about_ca_system_score_gemma":0.002319271,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006755732,"about_ca_topic_score_gemma":0.005668946,"domain_scores_codex":[0.9395158,0.0366964,0.012702,0.004449867,0.006007604,0.0006282932],"domain_scores_gemma":[0.4942804,0.3865929,0.07423792,0.01266207,0.03026854,0.001958186],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002141306,0.0005436318,0.9612357,0.002404038,0.003679118,0.0001386396,0.003029489,0.003610205,0.0002113018,0.0002375725,0.003419806,0.01934928],"study_design_scores_gemma":[0.001025793,0.001628775,0.9557157,0.001827697,0.002830014,0.0003627295,0.004285904,0.02465451,0.0008633952,0.0006814114,0.005946852,0.0001772573],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9701102,0.001092401,0.004795469,0.0004508074,0.00003928397,0.001674266,0.02055522,0.0001700882,0.001112337],"genre_scores_gemma":[0.9705357,0.0003711741,0.009318309,0.0002405094,0.00002704364,0.002351727,0.01690686,0.00005544968,0.0001932299],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9261263,"threshold_uncertainty_score":0.3906862,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0152825022894158,"score_gpt":0.3449879412991688,"score_spread":0.329705439009753,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}