{"id":"W2995031981","doi":"10.48550/arxiv.1912.05109","title":"Doubly Robust Off-Policy Actor-Critic Algorithms for Reinforcement Learning","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Estimator; Computer science; Variance (accounting); Function (biology); Bellman equation; Value (mathematics); Mathematical optimization; Artificial intelligence; Machine learning; Mathematics; Economics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003524678,0.001444366,0.001806555,0.0007378592,0.0004332028,0.001386418,0.001641284,0.001515952,0.002052184],"category_scores_gemma":[0.01383196,0.0007127197,0.000648469,0.0005713018,0.001581323,0.001231842,0.001779129,0.002318958,0.0005811196],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001141363,"about_ca_system_score_gemma":0.001149699,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002543537,"about_ca_topic_score_gemma":0.001716857,"domain_scores_codex":[0.9983754,0.0007090909,0.00008952006,0.0002645775,0.0004378985,0.0001235774],"domain_scores_gemma":[0.9941702,0.003848028,0.0005519278,0.0005244408,0.0007265278,0.0001789137],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006490528,0.00003451396,0.0004437749,0.00006799693,0.0000424325,0.00004012649,0.00004247509,0.9371818,0.0008164041,0.02396989,0.0006013922,0.03669429],"study_design_scores_gemma":[0.000003863312,0.00001269656,0.0000265773,0.00000489552,0.000002828401,0.000006070852,0.000001379364,0.9951338,0.0002394002,0.004366127,0.0001986534,0.000003658356],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003067806,0.0002158914,0.9952645,0.00008847607,0.00002771549,0.00002284718,0.00001394399,0.0002037289,0.001095021],"genre_scores_gemma":[0.7133296,0.0005299152,0.280081,0.0002476491,0.0001042623,0.0002753648,0.0001511696,0.000289755,0.004991218],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003524678,"threshold_uncertainty_score":0.01864052,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1024083187735273,"score_gpt":0.2271524172624324,"score_spread":0.1247440984889051,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}