{"id":"W4405635770","doi":"10.1136/bmj-2024-081948","title":"Age against the machine—susceptibility of large language models to cognitive impairment: cross sectional analysis","year":2024,"lang":"en","type":"article","venue":"BMJ","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Montreal Cognitive Assessment; Stroop effect; Cognition; Test (biology); Psychology; Cognitive psychology; Executive functions; Cognitive impairment; Psychiatry","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001246564,0.0004091792,0.00034591,0.001348471,0.0005909148,0.0007812134,0.0004469802,0.0007835537,0.003240955],"category_scores_gemma":[0.004652256,0.0003890023,0.0005078432,0.0007144422,0.0003641326,0.00110806,0.0007362475,0.0007996121,0.0009503519],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002175459,"about_ca_system_score_gemma":0.0002511475,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002758506,"about_ca_topic_score_gemma":0.00205176,"domain_scores_codex":[0.9993199,0.0001525726,0.00006827962,0.0002409742,0.0001283654,0.00008985258],"domain_scores_gemma":[0.9963818,0.0006249559,0.001703581,0.0003196342,0.000486031,0.0004838934],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001706428,0.00007959805,0.9986491,0.000005907797,0.00003269514,0.00007583306,0.0001296518,0.00002397676,0.00008597143,0.0000136775,0.00009941211,0.0006335799],"study_design_scores_gemma":[0.000005128096,0.0003651311,0.9987082,0.000005751297,0.00002713495,0.0003656542,0.0001749496,0.0001377944,0.00004983585,0.00002771097,0.0001285806,0.00000410755],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9989554,0.00008451713,0.00008232743,0.00002217466,0.000004147862,0.0000187351,0.0003202485,0.000005661879,0.0005067262],"genre_scores_gemma":[0.9992611,0.00003357825,0.00006316182,0.00002284278,0.000007718005,0.00002371127,0.0003207986,0.000003536324,0.0002635709],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003240955,"threshold_uncertainty_score":0.01084203,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1121805591650438,"score_gpt":0.4734118033067244,"score_spread":0.3612312441416807,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}