{"id":"W2095564494","doi":"10.1142/s0219525911002998","title":"AN EMPIRICAL STUDY OF POTENTIAL-BASED REWARD SHAPING AND ADVICE IN COMPLEX, MULTI-AGENT SYSTEMS","year":2011,"lang":"en","type":"article","venue":"Advances in Complex Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":79,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Context (archaeology); Reward system; Nash equilibrium; Domain (mathematical analysis); Artificial intelligence; Machine learning; Psychology; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005362243,0.0003629373,0.0004174017,0.0005797269,0.0005072165,0.0009380387,0.0009274998,0.00116578,0.002068613],"category_scores_gemma":[0.08769254,0.00027746,0.0002582823,0.0006335902,0.001829145,0.002509935,0.0009512882,0.00152195,0.0001440129],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008182767,"about_ca_system_score_gemma":0.0004905882,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003037824,"about_ca_topic_score_gemma":0.001890323,"domain_scores_codex":[0.9981856,0.001177953,0.00008606816,0.0001892663,0.0002668833,0.00009421562],"domain_scores_gemma":[0.9064002,0.08117667,0.005383215,0.004009358,0.00192751,0.001103018],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005191091,0.0007315437,0.09140047,0.0004738571,0.0002403865,0.0002860273,0.001165544,0.7839344,0.002666675,0.05420731,0.001623084,0.06275163],"study_design_scores_gemma":[0.00009211767,0.0003934834,0.03093326,0.00004656086,0.00003144073,0.0001454921,0.0003951828,0.9317333,0.001046916,0.03371331,0.001430551,0.00003843076],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9506266,0.0007417652,0.04305803,0.001047908,0.00001552334,0.00006967784,0.00009524567,0.0000837109,0.004261613],"genre_scores_gemma":[0.9964637,0.000082602,0.003181021,0.00002365926,0.000004576128,0.00001404654,0.00002817761,0.000006946979,0.0001952567],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005362243,"threshold_uncertainty_score":0.02835858,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1324506532271785,"score_gpt":0.3578127690604147,"score_spread":0.2253621158332362,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}