{"id":"W3118394271","doi":"10.1609/aaai.v35i11.17166","title":"Solving Common-Payoff Games with Approximate Policy Iteration","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Stochastic game; Scalability; Artificial intelligence; Common knowledge (logic); Scale (ratio); Mathematical optimization; Mathematical economics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002313377,0.001197675,0.001648936,0.0006204409,0.0006138362,0.001321635,0.001594702,0.001608789,0.002347057],"category_scores_gemma":[0.01075181,0.0007520452,0.0006743508,0.0005447314,0.002058221,0.001934078,0.002311067,0.002634974,0.0004301044],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001812391,"about_ca_system_score_gemma":0.002990687,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006078674,"about_ca_topic_score_gemma":0.007838199,"domain_scores_codex":[0.9987458,0.0005028577,0.00006186396,0.0002455393,0.0002577051,0.0001860982],"domain_scores_gemma":[0.9946477,0.004128562,0.000314041,0.0003688283,0.0002967997,0.0002440728],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007579671,0.00006756953,0.0007154936,0.00006056691,0.00004340481,0.00004600167,0.00009594943,0.918689,0.0003528968,0.05278856,0.0009772474,0.0260875],"study_design_scores_gemma":[0.0000137949,0.00001447268,0.00002649962,0.000004860173,0.000003168199,0.000005985622,0.000008366132,0.9759575,0.0001394891,0.02357583,0.0002468883,0.000003289291],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02691086,0.0001885736,0.966407,0.0004552162,0.00004073917,0.00008384718,0.0000398642,0.0004585335,0.005415401],"genre_scores_gemma":[0.7468053,0.0001755191,0.2479342,0.000305024,0.00005053474,0.0003469327,0.0001389921,0.0001250662,0.004118333],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006078674,"threshold_uncertainty_score":0.01314992,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04921359465533086,"score_gpt":0.2888286847180552,"score_spread":0.2396150900627244,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}