{"id":"W4389482477","doi":"10.54254/2753-8818/19/20230542","title":"An evaluation of reinforcement learning performance in the iterated prisoner’s dilemma","year":2023,"lang":"en","type":"article","venue":"Theoretical and Natural Science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"University of Cambridge","keywords":"Reinforcement learning; Dilemma; Iterated function; Prisoner's dilemma; Reinforcement; Computer science; Artificial neural network; Artificial intelligence; Psychology; Social psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0033451,0.000816157,0.0007883237,0.0004783937,0.0003356895,0.0006366331,0.0007820518,0.0009077642,0.0009667824],"category_scores_gemma":[0.009292775,0.0001718173,0.000268943,0.0002516046,0.0005251214,0.0007712796,0.0005860254,0.0007163399,0.0001568546],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007840127,"about_ca_system_score_gemma":0.0005592027,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003623624,"about_ca_topic_score_gemma":0.001943079,"domain_scores_codex":[0.9989033,0.0005034307,0.00009599067,0.0001548229,0.0002274108,0.0001150614],"domain_scores_gemma":[0.9943768,0.00369283,0.0005878732,0.0002830663,0.0006777836,0.0003817535],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001821326,0.002272564,0.01358093,0.0004435322,0.0003370568,0.000226121,0.0002977537,0.8819666,0.01053108,0.003682642,0.0009424937,0.08389793],"study_design_scores_gemma":[0.00007422816,0.001945797,0.00242826,0.00001467449,0.00002949631,0.00004025367,0.00005327535,0.9903127,0.003743532,0.001124592,0.00021222,0.00002094844],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9688687,0.0004240857,0.02543001,0.0001710499,0.00005887805,0.0001041535,0.00004910642,0.0002886915,0.00460524],"genre_scores_gemma":[0.9924469,0.00007353564,0.006822288,0.00002001946,0.00000432404,0.00004372602,0.00004111872,0.00001057258,0.0005373163],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003623624,"threshold_uncertainty_score":0.01769078,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01822272713279608,"score_gpt":0.2948606301005942,"score_spread":0.2766379029677981,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}