{"id":"W2945428699","doi":"10.65109/izyj7724","title":"Training Cooperative Agents for Multi-Agent Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"","topic":"Traffic control and management","field":"Engineering","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Scalability; Convergence (economics); Distributed computing; Artificial intelligence; Training (meteorology); Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002512395,0.0007205589,0.0008898133,0.0002965287,0.0006496264,0.0007502221,0.001583442,0.001007168,0.0021494],"category_scores_gemma":[0.006536261,0.0004798083,0.0003646041,0.0002582217,0.00155119,0.001138636,0.001603223,0.002042965,0.0004203467],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001061494,"about_ca_system_score_gemma":0.001521101,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003423085,"about_ca_topic_score_gemma":0.004023233,"domain_scores_codex":[0.9991249,0.0003649093,0.00003427783,0.0001907506,0.000178981,0.0001060492],"domain_scores_gemma":[0.9970154,0.001690892,0.0003366377,0.000440077,0.0003404627,0.0001765728],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005830735,0.00007269462,0.0005127274,0.00002932185,0.00002703312,0.0000454368,0.00008336498,0.9611664,0.001533091,0.01184016,0.0005228908,0.02410852],"study_design_scores_gemma":[0.000008935395,0.00001852865,0.00002543164,0.00000184745,0.00000218629,0.000003726053,0.000004894521,0.9964929,0.000356251,0.002872864,0.0002105543,0.000001968231],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01913314,0.00006896345,0.9785331,0.0001431101,0.00002777125,0.00004830784,0.000009706933,0.0005221797,0.001513716],"genre_scores_gemma":[0.8519871,0.00006410536,0.145093,0.000116334,0.00002987107,0.0002087847,0.00003371494,0.000087314,0.002379806],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003423085,"threshold_uncertainty_score":0.01328695,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04206833014354224,"score_gpt":0.2558452475613802,"score_spread":0.213776917417838,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}