{"id":"W4413213080","doi":"10.1109/tifs.2025.3595415","title":"Online Reward Poisoning in Reinforcement Learning With Convergence Guarantee","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Information Forensics and Security","topic":"Supply Chain and Inventory Management","field":"Business, Management and Accounting","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Huawei Technologies (Canada)","funders":"Natural Science Foundation of Guangdong Province; National Natural Science Foundation of China","keywords":"Reinforcement learning; Computer science; Convergence (economics); Reinforcement; Artificial intelligence; Computer security; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002049979,0.0001497491,0.0001473015,0.0004762399,0.0002351506,0.0001700205,0.0000848909,0.00005256599,0.00007490761],"category_scores_gemma":[0.000008031886,0.0001356825,0.00003360712,0.0004891331,0.00005642563,0.001605108,0.000005827356,0.0002521751,0.00002654073],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005140977,"about_ca_system_score_gemma":0.00001849184,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000374297,"about_ca_topic_score_gemma":0.000340387,"domain_scores_codex":[0.9991356,0.000007190835,0.0003400364,0.0001292029,0.000195807,0.0001922339],"domain_scores_gemma":[0.999623,0.00002024687,0.0001176597,0.0001248239,0.0001047729,0.000009515709],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001052579,0.0004063644,0.007055499,0.001630345,0.0002493209,0.00001344441,0.004922408,0.6029066,0.00002405897,0.2680232,0.00417175,0.1095445],"study_design_scores_gemma":[0.003651764,0.0001556145,0.002630846,0.0005695913,0.00009909396,0.000002634616,0.007051964,0.8051589,0.0002914957,0.006201904,0.1735755,0.000610634],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4752961,0.00003256418,0.4929629,0.001701961,0.001039971,0.0008061312,0.000007543191,0.0002164391,0.02793641],"genre_scores_gemma":[0.9971792,0.00007529724,0.0001094786,0.002335313,0.00002824538,0.0000248154,0.00003792587,0.000005548417,0.0002041449],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5218831,"threshold_uncertainty_score":0.5532971,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008356554127959817,"score_gpt":0.2087122874656386,"score_spread":0.2003557333376788,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}