{"id":"W4416962789","doi":"10.1109/pst65910.2025.11268866","title":"Cyber Threat Mitigation with Knowledge-Infused Reinforcement Learning and LLM-Guided Policies","year":2025,"lang":"en","type":"article","venue":"","topic":"Information and Cyber Security","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Defence Research and Development Canada; University of Regina; National Research Council Canada","funders":"","keywords":"Reinforcement learning; Leverage (statistics); Context (archaeology); Action (physics); Function (biology); Graph; Intelligent agent","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001180819,0.001481335,0.001043176,0.0006895873,0.0004062114,0.0009214534,0.001739397,0.001346067,0.002414824],"category_scores_gemma":[0.005223427,0.0005630431,0.0006411499,0.0003642274,0.00119161,0.001457894,0.001687307,0.002283091,0.0005234508],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001235972,"about_ca_system_score_gemma":0.002235563,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009551181,"about_ca_topic_score_gemma":0.007399627,"domain_scores_codex":[0.9992146,0.0002303775,0.00003813175,0.0002228457,0.0001594902,0.0001344732],"domain_scores_gemma":[0.9979489,0.001197656,0.0002664912,0.0002032835,0.0002370622,0.0001465205],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006703714,0.00009323344,0.0006499691,0.0000516438,0.00002868553,0.00007817863,0.00006839808,0.9497057,0.001335389,0.003536968,0.0008190468,0.04356566],"study_design_scores_gemma":[0.000009081442,0.00002097468,0.00004867206,0.000005103831,0.000004628545,0.000008334227,0.000005229965,0.997225,0.000346628,0.002092208,0.0002297787,0.000004358429],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03501479,0.0005125158,0.9565132,0.0005560574,0.00009088138,0.0001008299,0.00008049441,0.002785076,0.004346153],"genre_scores_gemma":[0.9043821,0.0001589539,0.09251525,0.000356581,0.00004607041,0.0001535609,0.0001473784,0.0001712094,0.002069034],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009551181,"threshold_uncertainty_score":0.01899117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0098515291908446,"score_gpt":0.2676955010088258,"score_spread":0.2578439718179812,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}