{"id":"W7134938218","doi":"10.1109/aiot66900.2025.00136","title":"A Comparative Evaluation of Teacher-Guided Reinforcement Learning Techniques for Autonomous Cyber Operations","year":2025,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal; Royal Military College of Canada","funders":"","keywords":"Reinforcement learning; Automation; Control (management); Key (lock); Action (physics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015916,0.0005433168,0.0006518342,0.0005031832,0.0002937378,0.0004777013,0.001041027,0.0008410252,0.001991],"category_scores_gemma":[0.006233651,0.0002022723,0.0002509718,0.0003322525,0.0004520021,0.0008293134,0.0006235719,0.0007988407,0.0003000938],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007221706,"about_ca_system_score_gemma":0.0008420929,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004843998,"about_ca_topic_score_gemma":0.004153318,"domain_scores_codex":[0.9991826,0.0003816249,0.00004193527,0.00009828551,0.0002367259,0.00005881906],"domain_scores_gemma":[0.9945953,0.003957861,0.0001944598,0.0003812837,0.0006995329,0.0001715155],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002490239,0.001816554,0.002186647,0.0006127525,0.0001421036,0.00005918965,0.000220599,0.5375456,0.009109633,0.00215975,0.001087247,0.4425697],"study_design_scores_gemma":[0.0001796028,0.001681273,0.001310672,0.00001820219,0.00003947148,0.00004158687,0.00005670396,0.9890423,0.005880494,0.0007542211,0.0009804231,0.00001503381],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7101893,0.001950457,0.2708489,0.0003561159,0.0001249981,0.0003208964,0.0001232814,0.002847773,0.01323839],"genre_scores_gemma":[0.9565486,0.0003173975,0.04112274,0.00002785709,0.00001247749,0.00006395333,0.00007550154,0.00008269375,0.001748772],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004843998,"threshold_uncertainty_score":0.009631634,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08454914817489433,"score_gpt":0.3828440244483338,"score_spread":0.2982948762734394,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}