{"id":"W4388824487","doi":"10.1038/s41562-023-01710-w","title":"The Turing test is not a good benchmark for thought in LLMs","year":2023,"lang":"en","type":"letter","venue":"Nature Human Behaviour","topic":"Computability, Logic, AI Algorithms","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"ca_institutions":"Canadian Institute for Advanced Research","funders":"","keywords":"Test (biology); Turing test; Benchmark (surveying); Psychology; Computer science; Artificial intelligence; Biology; Geography; Ecology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01139269,0.0004906668,0.001284601,0.0008949178,0.003683607,0.004815958,0.002141207,0.02191461,0.01027994],"category_scores_gemma":[0.07874684,0.0003685489,0.000845332,0.0004338337,0.01155835,0.01097072,0.003096678,0.03119202,0.008311288],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004642033,"about_ca_system_score_gemma":0.002797368,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002076204,"about_ca_topic_score_gemma":0.002916778,"domain_scores_codex":[0.9929245,0.002874996,0.0004603367,0.0007580456,0.002233436,0.000748721],"domain_scores_gemma":[0.9483201,0.0394531,0.001310924,0.003355656,0.004218865,0.003341341],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001704787,0.00005125973,0.0007505435,0.000114236,0.00003945715,0.0005726994,0.0002974429,0.0002695692,0.0002438825,0.2289636,0.7261519,0.04237506],"study_design_scores_gemma":[0.00008058826,0.00005299133,0.0004908752,0.0002020046,0.00001353675,0.0003326786,0.0003062694,0.001312049,0.0004666945,0.7114739,0.2852169,0.00005149273],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.001350793,0.001298032,0.002439683,0.9700015,0.007219343,0.000009938165,0.00006871337,0.0001207103,0.01749136],"genre_scores_gemma":[0.1051402,0.001864919,0.004706377,0.8287559,0.03451995,0.0001834116,0.0001806745,0.0002313195,0.02441735],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.02191461,"threshold_uncertainty_score":0.06025106,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02567321357612041,"score_gpt":0.2997814718697058,"score_spread":0.2741082582935854,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}