{"id":"W4210401180","doi":"10.1371/journal.pone.0262249","title":"Breaking the bonds of reinforcement: Effects of trial outcome, rule consistency and rule complexity against exploitable and unexploitable opponents","year":2022,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Experimental Behavioral Economics Studies","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University; University of Alberta","funders":"Osk. Huttusen säätiö; University of Sussex","keywords":"Reinforcement learning; Reinforcement; Exploit; Outcome (game theory); Consistency (knowledge bases); Computer science; Game theory; Simple (philosophy); Adversary; Artificial intelligence; Psychology; Mathematical economics; Social psychology; Computer security; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006227269,0.0001154726,0.0004281466,0.0000570444,0.000961288,0.000030866,0.0001982945,0.00002976936,0.00006810253],"category_scores_gemma":[0.0001330248,0.0001079974,0.00004530479,0.0001382719,0.0006384089,0.0001867735,0.0005255393,0.0001156725,0.000001297925],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001133332,"about_ca_system_score_gemma":0.00004798051,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002847157,"about_ca_topic_score_gemma":0.000159102,"domain_scores_codex":[0.9985906,0.0001635333,0.0004163761,0.0001977703,0.0003761362,0.0002555322],"domain_scores_gemma":[0.99918,0.0002440872,0.0002819315,0.00018156,0.00004843676,0.00006398895],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.004836402,0.008714539,0.4678143,0.001448183,0.001747889,0.00001768385,0.1125569,0.00006226112,0.2803209,0.1202488,0.0006074882,0.001624651],"study_design_scores_gemma":[0.1598198,0.01003006,0.05017668,0.001569688,0.003335329,0.000006476047,0.394944,0.001248039,0.3406702,0.03058707,0.003497963,0.004114643],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.986549,0.0008316714,8.769683e-7,0.0002779731,0.00009869051,0.0008705693,0.00003033521,0.00002110462,0.01131978],"genre_scores_gemma":[0.9986536,0.0002184861,0.000286787,0.000094584,0.00002769634,0.0001785666,0.00000816008,0.00001150907,0.0005206357],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4176376,"threshold_uncertainty_score":0.7393547,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1241823366200126,"score_gpt":0.3071874736431783,"score_spread":0.1830051370231658,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}