{"id":"W4417338215","doi":"10.1109/icmlc66258.2025.11280109","title":"Learning to Communicate in Multi-Agent Reinforcement Learning for Autonomous Cyber Defence","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Defence Research and Development Canada; Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Battle; Autonomous agent; Error-driven learning; Intelligent agent; Game theory; Reinforcement; Limit (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001563419,0.0007606518,0.0008933873,0.0002780617,0.0003487204,0.0006245191,0.0009225756,0.0009061858,0.001238866],"category_scores_gemma":[0.004040452,0.0003154895,0.0003536739,0.0002355532,0.001227018,0.0007648458,0.000985095,0.001210399,0.0001732337],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009256424,"about_ca_system_score_gemma":0.0009572351,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003525214,"about_ca_topic_score_gemma":0.002542053,"domain_scores_codex":[0.9993433,0.0003377828,0.00003141144,0.0001053757,0.00010736,0.00007471861],"domain_scores_gemma":[0.9982533,0.0011794,0.0001989869,0.00007641678,0.0001862071,0.0001056352],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004281073,0.00005251367,0.0003531809,0.00003503289,0.00002073651,0.00005127645,0.00005772362,0.9775263,0.000711584,0.01036575,0.0002297687,0.01055331],"study_design_scores_gemma":[0.00001137902,0.00002384296,0.00003369625,0.000002290414,0.000002581942,0.000004087031,0.000003439601,0.9957295,0.0001255581,0.003961153,0.0001001971,0.000002303834],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05148074,0.0003604415,0.9447092,0.0004164173,0.000046948,0.00006105494,0.0000179675,0.0001961624,0.00271097],"genre_scores_gemma":[0.9547726,0.0001348082,0.04279077,0.0001025449,0.00002676251,0.000144868,0.00002036355,0.00002470562,0.001982559],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003525214,"threshold_uncertainty_score":0.008268237,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04737860407282586,"score_gpt":0.3158628626450966,"score_spread":0.2684842585722707,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}