{"id":"W4376223516","doi":"10.1613/jair.1.14580","title":"Exploiting Action Impact Regularity and Exogenous State Variables for Offline Reinforcement Learning","year":2023,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Component (thermodynamics); Exploit; Property (philosophy); Offline learning; Key (lock); Class (philosophy); Artificial intelligence; Machine learning; Action (physics); State (computer science); Reinforcement; Online and offline; State action; Online learning; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003654163,0.001088913,0.001606072,0.0005058318,0.0004875745,0.001014448,0.001512207,0.001315643,0.002520514],"category_scores_gemma":[0.0181138,0.0007169983,0.0006300849,0.000471616,0.002167293,0.002341434,0.001898147,0.003248899,0.0003927452],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001163698,"about_ca_system_score_gemma":0.002585687,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003555306,"about_ca_topic_score_gemma":0.003293181,"domain_scores_codex":[0.9988194,0.0004307451,0.00007712655,0.0002884094,0.0002259253,0.0001583231],"domain_scores_gemma":[0.9845781,0.01207999,0.001065993,0.001322389,0.0005153754,0.000438267],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001458166,0.00011928,0.001552982,0.00006064169,0.00002929687,0.0000702715,0.00006746766,0.9516208,0.0009576736,0.02027378,0.0004639344,0.0246379],"study_design_scores_gemma":[0.00001377154,0.00002979271,0.00007534039,0.000004829109,0.000002608174,0.000008989831,0.000004003845,0.9908174,0.0002989031,0.008624328,0.0001168632,0.000003237892],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0266654,0.0000919423,0.9711018,0.0002121995,0.00001993349,0.00007719309,0.00004024864,0.0004272982,0.001364039],"genre_scores_gemma":[0.8793874,0.0001031887,0.1185985,0.0001872398,0.00003686561,0.0001990334,0.0001340918,0.0001176328,0.00123605],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003654163,"threshold_uncertainty_score":0.01932532,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2947350345956782,"score_gpt":0.4630904091938908,"score_spread":0.1683553745982125,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}