{"id":"W4403024005","doi":"10.1109/iccims61672.2024.10690552","title":"Training Environments for Reinforcement Learning Cybersecurity Agents","year":2024,"lang":"en","type":"article","venue":"","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Defence Research and Development Canada; Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Computer science; Computer security; Training (meteorology); Reinforcement; Human–computer interaction; Artificial intelligence; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001104671,0.0005652415,0.0004216357,0.0002740209,0.0003824266,0.0008928359,0.001363459,0.0009267486,0.005209385],"category_scores_gemma":[0.005391927,0.0003786343,0.0003223265,0.0001424501,0.0008712171,0.001221789,0.001810562,0.001471692,0.0006893537],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006170701,"about_ca_system_score_gemma":0.0007999853,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009270961,"about_ca_topic_score_gemma":0.001066676,"domain_scores_codex":[0.9992337,0.0004128253,0.00004494982,0.00009731251,0.0001432863,0.00006788722],"domain_scores_gemma":[0.9975656,0.001682132,0.0002006841,0.0001661445,0.0002185405,0.000166752],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001617175,0.0002238152,0.0006732569,0.0001520129,0.00002206184,0.0001119228,0.0002000141,0.8929725,0.0035066,0.03498011,0.001332768,0.06566318],"study_design_scores_gemma":[0.00006596724,0.000217691,0.0001582191,0.00006456196,0.000007121303,0.00004998882,0.0000533919,0.9750695,0.002659867,0.01442313,0.007217017,0.00001354893],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02742862,0.0005323127,0.9620182,0.0004726405,0.00007728041,0.0001633669,0.00004387109,0.001113998,0.008149779],"genre_scores_gemma":[0.6635581,0.0008967101,0.3296513,0.0002835623,0.00007058978,0.0006365568,0.0001420054,0.0001545686,0.004606675],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005209385,"threshold_uncertainty_score":0.01742715,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03925782756135474,"score_gpt":0.3006183886608535,"score_spread":0.2613605610994987,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}