{"id":"W6910537080","doi":"10.48620/85260","title":"A comparison of large language model-generated and published perioperative neurocognitive disorder recommendations: a cross-sectional web-based analysis","year":2025,"lang":"en","type":"article","venue":"Open Access CRIS of the University of Bern","topic":"Intensive Care Unit Cognitive Disorders","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Neurocognitive; Perioperative; Reliability (semiconductor); MEDLINE; Quality (philosophy); Patient safety; Quality of life (healthcare)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002024958,0.0001274259,0.0005033743,0.0004525947,0.000240104,0.00008695928,0.0006130667,0.00007388632,0.0005402958],"category_scores_gemma":[0.001055414,0.0001156425,0.0001998286,0.001189759,0.0003733626,0.0005321531,0.0007739126,0.0001886706,5.905517e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000460009,"about_ca_system_score_gemma":0.0003073876,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001707021,"about_ca_topic_score_gemma":0.002074199,"domain_scores_codex":[0.9989709,0.0001645195,0.0002695929,0.0002786905,0.0001935994,0.0001227231],"domain_scores_gemma":[0.9939325,0.000143715,0.0003038647,0.0002609524,0.005319674,0.00003932243],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000960961,0.0005126246,0.9883591,0.00009121648,0.001105595,0.000001726094,0.001837841,0.001866222,0.002637522,0.0002399126,0.002065493,0.0003217695],"study_design_scores_gemma":[0.006578939,0.0001205701,0.8041049,0.0002087737,0.001722011,5.427562e-7,0.02736319,0.1450247,0.01422477,0.00004075735,0.000434803,0.0001760669],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9818613,0.00007831732,0.006346492,0.00281313,0.00004370429,0.0006589503,0.0006363257,0.000009159903,0.007552619],"genre_scores_gemma":[0.9963936,0.00001429591,0.0004299003,0.001152581,0.000002310457,0.000001873444,0.0002810915,0.000007147024,0.001717249],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1842543,"threshold_uncertainty_score":0.591586,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03503860180602209,"score_gpt":0.3987508128312685,"score_spread":0.3637122110252464,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}