{"id":"W4221102161","doi":"10.1007/s11227-022-04305-w","title":"Improved reinforcement learning in cooperative multi-agent environments using knowledge transfer","year":2022,"lang":"en","type":"article","venue":"The Journal of Supercomputing","topic":"Distributed Control Multi-Agent Systems","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Reinforcement learning; Abstraction; State space; Convergence (economics); Process (computing); Artificial intelligence; Fuse (electrical); Mechanism (biology); State (computer science); Space (punctuation); Transfer of learning; Class (philosophy); Machine learning; Distributed computing; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00192961,0.0007193143,0.001278055,0.000427862,0.0005932958,0.0008374757,0.001787978,0.001326337,0.001924345],"category_scores_gemma":[0.006922085,0.0004146144,0.0003899803,0.0003506112,0.001142018,0.001495474,0.002048835,0.001273023,0.0002859873],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007242679,"about_ca_system_score_gemma":0.001004123,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0045069,"about_ca_topic_score_gemma":0.002882598,"domain_scores_codex":[0.9992418,0.0003070625,0.00003967471,0.0001178635,0.0001726525,0.0001209751],"domain_scores_gemma":[0.9964517,0.002377052,0.000213024,0.0002864425,0.0004877968,0.000183959],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001539917,0.0001497533,0.0005068327,0.00004088233,0.00004244134,0.00006544545,0.00008660057,0.9622668,0.001719689,0.005164966,0.0003546824,0.02944792],"study_design_scores_gemma":[0.00001389786,0.00002417182,0.00003818144,0.000001334742,0.000003897464,0.000004449837,0.000003499963,0.9983643,0.0001917343,0.00130732,0.00004486092,0.000002411054],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.124403,0.0002433872,0.8696931,0.0003166032,0.000081199,0.00008435094,0.00001939481,0.0004178282,0.004741197],"genre_scores_gemma":[0.9688324,0.00004795765,0.02946234,0.00005181731,0.00002286513,0.00006485156,0.00001762763,0.00002475386,0.001475386],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0045069,"threshold_uncertainty_score":0.01020485,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03410398160610389,"score_gpt":0.2610866262253376,"score_spread":0.2269826446192337,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}