{"id":"W4403598347","doi":"10.48550/arxiv.2409.04792","title":"Improving Deep Reinforcement Learning by Reducing the Chain Effect of Value and Policy Churn","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Transportation and Mobility Innovations","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; Fonds Québécois de la Recherche sur la Nature et les Technologies; Nvidia","keywords":"Reinforcement learning; Reinforcement; Value (mathematics); Chain (unit); Computer science; Learning effect; Artificial intelligence; Microeconomics; Economics; Machine learning; Psychology; Social psychology; Physics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002720704,0.0001917419,0.0001967175,0.0002109478,0.00008695484,0.00003013143,0.0001656519,0.0001327479,0.00001284889],"category_scores_gemma":[0.00003913617,0.000179263,0.0000879845,0.000395532,0.0000720827,0.00004784428,0.0001320358,0.0006370794,0.00000439435],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001301732,"about_ca_system_score_gemma":0.00004703625,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006682739,"about_ca_topic_score_gemma":0.00002567744,"domain_scores_codex":[0.9992569,0.00004315282,0.0001988703,0.0002792481,0.00005208365,0.000169753],"domain_scores_gemma":[0.999482,0.00008999872,0.0000754303,0.0002677315,0.00003754893,0.00004730427],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008632173,0.000003511172,0.0002836543,0.0006871722,0.0001089454,0.000004674338,0.0006602091,0.9742312,0.002081051,0.02049867,0.00003396643,0.001398259],"study_design_scores_gemma":[0.0002832849,0.00005452337,0.0003648976,0.0001692517,0.0001820777,0.000001189332,0.0002135643,0.9944587,0.003042152,0.0007466666,0.0002677951,0.000215884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9386873,0.0002429187,0.05797698,0.0000589914,0.000220361,0.0003773349,0.00001105489,0.0002442645,0.002180754],"genre_scores_gemma":[0.9990953,0.0001565593,0.00002215597,0.00001160179,0.00004258488,0.000002684162,0.00004399316,0.00002549011,0.0005996891],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0604079,"threshold_uncertainty_score":0.7310134,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01319784069502119,"score_gpt":0.1746416961651146,"score_spread":0.1614438554700934,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}