{"id":"W2180467047","doi":"10.1007/s10462-015-9447-5","title":"Exponential moving average based multiagent reinforcement learning algorithms","year":2015,"lang":"en","type":"article","venue":"Artificial Intelligence Review","topic":"Adaptive Dynamic Programming Control","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University","funders":"","keywords":"Nash equilibrium; Computer science; Reinforcement learning; Algorithm; Convergence (economics); Q-learning; Weighted Majority Algorithm; Mathematical optimization; Artificial intelligence; Mathematics; Wake-sleep algorithm; Unsupervised learning; Generalization error","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001475053,0.0007967117,0.001122087,0.0005754697,0.0002724277,0.0008476349,0.001920846,0.0009036099,0.002058991],"category_scores_gemma":[0.003549219,0.0002840502,0.0003919424,0.0007205952,0.0005467906,0.001153807,0.0008146957,0.001376671,0.0003743491],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006077897,"about_ca_system_score_gemma":0.0005448267,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00214475,"about_ca_topic_score_gemma":0.001733788,"domain_scores_codex":[0.9994926,0.0001822362,0.00003299732,0.00007486495,0.0001803567,0.00003683905],"domain_scores_gemma":[0.9984549,0.001039564,0.00009548625,0.00006650971,0.0003057483,0.00003774615],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005878008,0.0001044633,0.0004362051,0.0001543401,0.00008809959,0.00003109443,0.000036358,0.7085257,0.0004903006,0.03254167,0.002216151,0.2553168],"study_design_scores_gemma":[0.00001411389,0.00003548747,0.0001020659,0.00001815832,0.00001465821,0.0000225797,0.000004998727,0.9886931,0.0002846867,0.009048709,0.001754977,0.000006516101],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008172801,0.005466222,0.978992,0.0003517083,0.0002119652,0.00003162206,0.00001766059,0.0002510456,0.006504895],"genre_scores_gemma":[0.6313337,0.009593002,0.3414608,0.0004852742,0.0004104115,0.0002697038,0.0001321929,0.0001442199,0.01617068],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00214475,"threshold_uncertainty_score":0.007800937,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0888835926429212,"score_gpt":0.3256896554819136,"score_spread":0.2368060628389924,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}