{"id":"W4319301677","doi":"10.1162/neco_a_01563","title":"Large Language Models and the Reverse Turing Test","year":2023,"lang":"en","type":"article","venue":"Neural Computation","topic":"Topic Modeling","field":"Computer Science","cited_by":164,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute of Biomedical Imaging and Bioengineering; Canadian Institute for Advanced Research; National Science Foundation","keywords":"Psychology; Cognitive psychology; Computer science; Cognitive science","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03664291,0.001363994,0.002484507,0.001795643,0.002505296,0.006396725,0.003272119,0.003727553,0.007131889],"category_scores_gemma":[0.1961903,0.0008982833,0.002500898,0.001046843,0.01282162,0.01532806,0.007765543,0.009220462,0.002043824],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003094039,"about_ca_system_score_gemma":0.002615485,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002417831,"about_ca_topic_score_gemma":0.001422441,"domain_scores_codex":[0.9473379,0.04072325,0.001906686,0.004224399,0.004491286,0.001316424],"domain_scores_gemma":[0.7187572,0.2408734,0.004840067,0.02371841,0.00903827,0.002772656],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003958105,0.0001187722,0.004687582,0.0006986887,0.0003030736,0.0007374256,0.002436507,0.01953303,0.0005974481,0.8816514,0.02116675,0.0676735],"study_design_scores_gemma":[0.00005404389,0.00004180288,0.0004858665,0.00009779973,0.00002527448,0.0001632068,0.0002329066,0.03448628,0.0005335563,0.9568704,0.00695741,0.0000514273],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.163402,0.01199325,0.6068138,0.1024332,0.00252898,0.0006571512,0.001765667,0.004003155,0.1064029],"genre_scores_gemma":[0.8244039,0.002155496,0.1466707,0.008717726,0.001799412,0.00129062,0.001861537,0.001005766,0.01209466],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03664291,"threshold_uncertainty_score":0.1937885,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03063292580025428,"score_gpt":0.2742570784898651,"score_spread":0.2436241526896109,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}