{"id":"W2808117931","doi":"10.65109/ohiq4340","title":"Faster Policy Adaptation in Environments with Exogeneity: A State Augmentation Approach","year":2018,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Kernel (algebra); State (computer science); Embedding; Subspace topology; Variable (mathematics); State space; Function (biology); Q-learning; Variance (accounting); Artificial intelligence; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001433183,0.0000978328,0.00007360995,0.0001719914,0.00005110089,0.00009976905,0.0002687126,0.00002345666,0.00001403957],"category_scores_gemma":[0.000007390804,0.0000805563,0.00001118586,0.0003351019,0.00005170211,0.0005716097,0.0000999246,0.00005844949,0.0001138747],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001232981,"about_ca_system_score_gemma":0.0000485707,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001311904,"about_ca_topic_score_gemma":0.00003222259,"domain_scores_codex":[0.9990174,0.00004857088,0.000165893,0.0002546151,0.0002984756,0.0002150616],"domain_scores_gemma":[0.9995673,0.00001340846,0.00007785548,0.0002816197,0.00001494515,0.00004490548],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002725875,0.00006525649,0.006534345,0.000008715976,0.00001921238,0.000003008404,0.008119457,0.9522208,0.0007366533,0.006983546,0.00004506662,0.02523669],"study_design_scores_gemma":[0.0006134509,0.000254659,0.01562018,0.000007141574,0.000001892438,0.000003804409,0.0001614095,0.9811161,0.001450245,0.0001435987,0.0004953763,0.0001322053],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02739912,0.00000152947,0.9615623,0.00013764,0.00003262379,0.000190743,1.729338e-7,0.00004199352,0.0106339],"genre_scores_gemma":[0.8206172,0.00000318836,0.1756842,0.0003567966,0.00003169376,0.00001496044,0.000005379401,0.000007481749,0.003279153],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7932181,"threshold_uncertainty_score":0.3284991,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02230516015929545,"score_gpt":0.2434765207205141,"score_spread":0.2211713605612186,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}