{"id":"W2945428699","doi":"10.65109/izyj7724","title":"Training Cooperative Agents for Multi-Agent Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"","topic":"Traffic control and management","field":"Engineering","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Scalability; Convergence (economics); Distributed computing; Artificial intelligence; Training (meteorology); Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00009669316,0.0001118822,0.0001295797,0.00004109638,0.00004211712,0.00002771218,0.00006905078,0.00002552864,0.0004577414],"category_scores_gemma":[0.000008833449,0.00009996734,0.00005445839,0.00004284285,0.000004341679,0.00006735597,0.00002075588,0.00006675826,0.0001921242],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004718742,"about_ca_system_score_gemma":0.000005946198,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00000428045,"about_ca_topic_score_gemma":0.00001156958,"domain_scores_codex":[0.9994053,0.00000592987,0.0001483973,0.0001269074,0.0000813593,0.0002321511],"domain_scores_gemma":[0.9998085,0.00002090987,0.00001342604,0.00009000489,0.00002103182,0.00004614086],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006363826,0.000007503173,0.00002523188,0.00003963078,0.00008345592,6.436356e-7,0.001103736,0.9647524,0.0006626793,0.002175405,0.001050777,0.03009215],"study_design_scores_gemma":[0.001231982,0.00006692413,0.0003857927,0.00001165601,0.00001122349,2.125533e-7,0.0008399238,0.7781045,0.0001402357,0.000001401977,0.2190769,0.0001292021],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05770418,0.0001006938,0.8604769,0.00007542368,0.001102126,0.001799848,0.000002194279,0.000968469,0.07777023],"genre_scores_gemma":[0.9639418,0.00001603869,0.002205012,0.0001115572,0.00003397243,0.00009723459,0.00001428481,0.00002092407,0.03355924],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9062375,"threshold_uncertainty_score":0.5011947,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04206833014354224,"score_gpt":0.2558452475613802,"score_spread":0.213776917417838,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}