{"id":"W4416962789","doi":"10.1109/pst65910.2025.11268866","title":"Cyber Threat Mitigation with Knowledge-Infused Reinforcement Learning and LLM-Guided Policies","year":2025,"lang":"en","type":"article","venue":"","topic":"Information and Cyber Security","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Defence Research and Development Canada; University of Regina; National Research Council Canada","funders":"","keywords":"Reinforcement learning; Leverage (statistics); Context (archaeology); Action (physics); Function (biology); Graph; Intelligent agent","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001495195,0.00009634504,0.0000929458,0.000124469,0.000204932,0.0002062739,0.0001655568,0.00003609842,0.00003151883],"category_scores_gemma":[0.00001559818,0.0000725652,0.00001879919,0.0003179833,0.00004062482,0.0005721981,0.0001428639,0.00008696802,0.00003638637],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003599105,"about_ca_system_score_gemma":0.00007460053,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000084026,"about_ca_topic_score_gemma":0.00006227744,"domain_scores_codex":[0.9993809,0.0000240325,0.0001799197,0.0001335775,0.0001236555,0.0001579013],"domain_scores_gemma":[0.9995804,0.00003394245,0.00004866123,0.0001702207,0.0001213308,0.00004541088],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004393363,0.00001821827,0.002591813,0.00002745702,0.0000202388,5.064998e-7,0.006259602,0.0003807296,0.00005389751,0.9794723,0.002817206,0.008353639],"study_design_scores_gemma":[0.004372073,0.0004526577,0.03620477,0.0002856017,0.00003960852,0.00003810905,0.002363729,0.6897881,0.02949422,0.007457805,0.2285639,0.0009394077],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.03356776,0.00005521964,0.3083542,0.001614895,0.00009503327,0.0001759828,9.911979e-8,0.0002511517,0.6558857],"genre_scores_gemma":[0.9786281,0.00001973992,0.003772859,0.0009842939,0.00001227403,0.00001295468,0.00000371706,0.000002453872,0.0165636],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9720145,"threshold_uncertainty_score":0.2959123,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0098515291908446,"score_gpt":0.2676955010088258,"score_spread":0.2578439718179812,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}