{"id":"W2901559512","doi":"10.22215/etd/2013-09907","title":"Study of Multiple Multiagent Reinforcement Learning Algorithms in Grid Games","year":2013,"lang":"en","type":"dissertation","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Nash equilibrium; Minimax; Computer science; Algorithm; Best response; Grid; Game theory; Mathematical optimization; Artificial intelligence; Mathematics; Mathematical economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004166694,0.0004274772,0.0005994458,0.0005936629,0.0000931606,0.0001593593,0.001303732,0.0002165309,0.0001471628],"category_scores_gemma":[0.000237579,0.0003982933,0.0001152808,0.0005065833,0.0000168674,0.000435836,0.0002893893,0.0007150832,0.0001221341],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001372478,"about_ca_system_score_gemma":0.0001091062,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001576156,"about_ca_topic_score_gemma":0.0003034911,"domain_scores_codex":[0.9965774,0.0001559481,0.001148135,0.0006631042,0.0009888493,0.0004665956],"domain_scores_gemma":[0.9978614,0.000203542,0.0007630153,0.0008167829,0.0002592554,0.00009602866],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001073268,0.0002120263,0.004196857,0.0001002217,0.0000731703,0.00001515647,0.01334395,0.9747301,0.0001083239,0.0001819964,0.0003068509,0.006720598],"study_design_scores_gemma":[0.00163399,0.0009374625,0.01733188,0.0001811663,0.00002214206,0.000001110375,0.0112661,0.965827,0.00159958,0.0000041576,0.0006856467,0.0005098048],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3828625,0.0001689868,0.5867743,0.00002726567,0.005268916,0.004689835,5.906398e-7,0.0004868327,0.01972082],"genre_scores_gemma":[0.9329322,0.00008042626,0.01346196,0.00002497023,0.0001028979,0.0002276126,0.0001929123,0.00004828452,0.05292873],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5733123,"threshold_uncertainty_score":0.9998469,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02480748313668581,"score_gpt":0.2822074256128413,"score_spread":0.2573999424761556,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}