{"id":"W3085332162","doi":"10.22148/001c.17212","title":"Can GPT-3 Pass a Writer’s Turing Test?","year":2020,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Computability, Logic, AI Algorithms","field":"Computer Science","cited_by":205,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Turing test; Heuristic; Field (mathematics); Grammar; Test (biology); Artificial intelligence; Scale (ratio); Language model; Natural language processing; Natural language; Turing; Language identification; Programming language; Linguistics; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02406939,0.001593007,0.001454692,0.001354577,0.004158891,0.007674352,0.003408726,0.004879539,0.02701957],"category_scores_gemma":[0.1270535,0.000697776,0.001777639,0.001220152,0.007660642,0.01798639,0.005734114,0.008590098,0.03745231],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002433242,"about_ca_system_score_gemma":0.005002933,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002795217,"about_ca_topic_score_gemma":0.002339252,"domain_scores_codex":[0.9858576,0.00573923,0.0007556164,0.002436464,0.003993981,0.001217148],"domain_scores_gemma":[0.9410293,0.03246045,0.001184489,0.01137457,0.01095422,0.002997007],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003141781,0.0001318508,0.00160174,0.0003354937,0.00009748532,0.0002603903,0.0007503344,0.00217276,0.0005613537,0.2166682,0.5968908,0.1802154],"study_design_scores_gemma":[0.0001388323,0.0002251583,0.0008551439,0.0004568344,0.00005579696,0.0004934738,0.0004926056,0.009401583,0.002623856,0.4408433,0.5442974,0.0001161065],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.02145346,0.01379599,0.2102687,0.5001205,0.02971779,0.0005170859,0.003859173,0.01559705,0.2046701],"genre_scores_gemma":[0.3295205,0.01292814,0.3311094,0.1180474,0.01346921,0.002467877,0.009113978,0.01431151,0.169032],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02701957,"threshold_uncertainty_score":0.1272926,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04380876233099833,"score_gpt":0.2608145662498129,"score_spread":0.2170058039188145,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}