{"id":"W4376223516","doi":"10.1613/jair.1.14580","title":"Exploiting Action Impact Regularity and Exogenous State Variables for Offline Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Component (thermodynamics); Exploit; Property (philosophy); Offline learning; Key (lock); Class (philosophy); Artificial intelligence; Machine learning; Action (physics); State (computer science); Reinforcement; Online and offline; State action; Online learning; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009659824,0.0001491343,0.0002753249,0.0008389419,0.0006445567,0.0005938023,0.0006919449,0.00007998419,0.00002109842],"category_scores_gemma":[0.002811672,0.0001287074,0.0001321505,0.001287497,0.0001256503,0.0008916739,0.0003587169,0.0008462504,0.00003262782],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002420264,"about_ca_system_score_gemma":0.0003538869,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006657487,"about_ca_topic_score_gemma":0.000005557668,"domain_scores_codex":[0.9965653,0.0002922209,0.0009119301,0.0002642747,0.001215718,0.0007505437],"domain_scores_gemma":[0.9963825,0.00132169,0.0004323517,0.000274738,0.001364103,0.0002245645],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001068271,0.00002196254,0.0001424084,0.00005367651,0.00005034784,0.0000294422,0.001596597,0.8649825,0.01283599,0.005538122,0.0001733563,0.1144688],"study_design_scores_gemma":[0.00006768596,0.001453591,0.0001662858,0.0001026444,0.000006904472,0.00004984747,0.001270963,0.9528758,0.02766304,0.01535973,0.0008482547,0.0001352519],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08895168,0.00006037295,0.909827,0.0004505526,0.0002783884,0.0002748818,5.42821e-7,0.00005594893,0.0001006656],"genre_scores_gemma":[0.9770989,0.0006508591,0.02125621,0.00001376352,0.000339159,0.00001116862,0.000003510467,0.00002187645,0.0006045336],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8885708,"threshold_uncertainty_score":0.5726048,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2947350345956782,"score_gpt":0.4630904091938908,"score_spread":0.1683553745982125,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}