{"id":"W3037719421","doi":"10.65109/gjmw6851","title":"Neural Replicator Dynamics: Multiagent Learning via Hedging Policy Gradients","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Softmax function; Reinforcement learning; Computer science; Nash equilibrium; Mathematical optimization; Replicator equation; Convergence (economics); Gradient descent; Best response; Regret; Margin (machine learning); Artificial intelligence; Artificial neural network; Applied mathematics; Mathematical economics; Mathematics; Machine learning; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001769704,0.0008395531,0.001003726,0.000363666,0.000362438,0.0009510462,0.001673836,0.001205801,0.002362369],"category_scores_gemma":[0.005453129,0.0005747338,0.000418832,0.0003236805,0.001103723,0.001515032,0.001548283,0.002043881,0.0003920272],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009384217,"about_ca_system_score_gemma":0.001028395,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003215012,"about_ca_topic_score_gemma":0.002869691,"domain_scores_codex":[0.9995891,0.0001722135,0.0000204726,0.00009418538,0.00008422016,0.00003974051],"domain_scores_gemma":[0.9984351,0.001058624,0.0001562508,0.0001354231,0.0001316245,0.00008298688],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006081434,0.00005208438,0.0008199678,0.00004652288,0.00005285395,0.00006853764,0.00008403134,0.9206044,0.0009077126,0.028983,0.001050461,0.04726966],"study_design_scores_gemma":[0.000006464481,0.00001041615,0.00002281649,0.000003217388,0.000002320914,0.000006134893,0.000002578893,0.994585,0.0001503689,0.005054641,0.0001540082,0.00000214578],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02724915,0.000317453,0.9686906,0.0003626921,0.00005651744,0.00006435017,0.00003270885,0.0004762481,0.002750332],"genre_scores_gemma":[0.8394433,0.0002297276,0.1557322,0.0002557369,0.00003779022,0.0002027006,0.00007134928,0.00009761903,0.003929529],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003215012,"threshold_uncertainty_score":0.009359241,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01798073952334079,"score_gpt":0.2518673730577505,"score_spread":0.2338866335344097,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}