{"id":"W4413213080","doi":"10.1109/tifs.2025.3595415","title":"Online Reward Poisoning in Reinforcement Learning With Convergence Guarantee","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Information Forensics and Security","topic":"Supply Chain and Inventory Management","field":"Business, Management and Accounting","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Huawei Technologies (Canada)","funders":"Natural Science Foundation of Guangdong Province; National Natural Science Foundation of China","keywords":"Reinforcement learning; Computer science; Convergence (economics); Reinforcement; Artificial intelligence; Computer security; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004677458,0.00135455,0.001907084,0.000545062,0.000649775,0.001317941,0.001985088,0.002064034,0.002819843],"category_scores_gemma":[0.01774871,0.0006335606,0.0005083731,0.0005064785,0.002182919,0.002094307,0.002230646,0.003301253,0.0006640095],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001618268,"about_ca_system_score_gemma":0.00247969,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002617378,"about_ca_topic_score_gemma":0.001624194,"domain_scores_codex":[0.9980982,0.0007696674,0.00008719358,0.0003857267,0.00039125,0.0002679458],"domain_scores_gemma":[0.9895239,0.008127934,0.0006488357,0.0006027207,0.0007319122,0.0003646302],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002759031,0.0001340335,0.001182665,0.0001427503,0.00004130592,0.0001384649,0.0001138327,0.9230765,0.0009580938,0.02660146,0.001429951,0.04590491],"study_design_scores_gemma":[0.00002613304,0.00004239207,0.0000408061,0.000009542251,0.000003991277,0.00002233547,0.000005587549,0.9903055,0.0002872819,0.009060589,0.0001923721,0.000003385954],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02320155,0.0005925269,0.971193,0.0005561747,0.00005093787,0.0001085376,0.00004315816,0.001149914,0.003104117],"genre_scores_gemma":[0.8901171,0.0003023035,0.105773,0.0003148842,0.0000730168,0.0002050294,0.00009068405,0.0001701424,0.00295388],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004677458,"threshold_uncertainty_score":0.02473706,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008356554127959817,"score_gpt":0.2087122874656386,"score_spread":0.2003557333376788,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}