{"id":"W4417338215","doi":"10.1109/icmlc66258.2025.11280109","title":"Learning to Communicate in Multi-Agent Reinforcement Learning for Autonomous Cyber Defence","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Defence Research and Development Canada; Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Battle; Autonomous agent; Error-driven learning; Intelligent agent; Game theory; Reinforcement; Limit (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.003025356,0.0008295411,0.0009464354,0.00115815,0.001272153,0.001108744,0.003851706,0.0003565047,0.000168737],"category_scores_gemma":[0.001824897,0.0009287493,0.000362426,0.002266018,0.0001839901,0.0007455299,0.004552958,0.002013079,0.000504596],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00126771,"about_ca_system_score_gemma":0.0007059305,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005505336,"about_ca_topic_score_gemma":0.0001371885,"domain_scores_codex":[0.9929937,0.0006498458,0.002177013,0.001487046,0.0007113567,0.001981109],"domain_scores_gemma":[0.9954557,0.001138709,0.0006052031,0.001938585,0.0004628344,0.0003989958],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007075955,0.0001144592,0.006681448,0.0001837763,0.0001034026,0.000009627306,0.007071712,0.9461347,0.0002702565,0.0112866,0.0003445811,0.02772865],"study_design_scores_gemma":[0.002402449,0.0009225266,0.002413886,0.0007654966,0.00004708575,0.000002803726,0.001079384,0.9194852,0.0007305978,0.00003817658,0.07125571,0.0008566558],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002905538,0.0002795141,0.9729749,0.002750093,0.001233101,0.002858126,3.907667e-7,0.0003938825,0.01660447],"genre_scores_gemma":[0.758519,0.0002353451,0.1129518,0.001769834,0.00002998921,0.0003171676,0.00001496503,0.00005156862,0.1261103],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8600231,"threshold_uncertainty_score":0.9999282,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04737860407282586,"score_gpt":0.3158628626450966,"score_spread":0.2684842585722707,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}