{"id":"W2796389682","doi":"10.1109/cdc.2018.8619180","title":"Renewal Monte Carlo: Renewal Theory Based Reinforcement Learning","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Markov decision process; Monte Carlo method; Estimator; Computer science; Reinforcement learning; Mathematical optimization; Importance sampling; Variance (accounting); Key (lock); Markov process; Mathematics; Artificial intelligence; Statistics; Economics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002039044,0.0008558602,0.001150298,0.0007940037,0.0004905066,0.001750836,0.001931485,0.001462358,0.006458944],"category_scores_gemma":[0.007852039,0.0005381095,0.0007184631,0.0008724486,0.00113438,0.001591734,0.001572328,0.002277542,0.001477816],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001538581,"about_ca_system_score_gemma":0.002395795,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005719583,"about_ca_topic_score_gemma":0.004128654,"domain_scores_codex":[0.9984694,0.0005679768,0.00005694067,0.0002501267,0.0005110495,0.000144429],"domain_scores_gemma":[0.9972864,0.001756708,0.0001988429,0.0002357929,0.0003584404,0.0001636728],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001534606,0.000151348,0.0009012585,0.0001219999,0.00005963017,0.0001251986,0.00006862944,0.7400078,0.001234522,0.09773013,0.006782459,0.1526636],"study_design_scores_gemma":[0.0000141478,0.00001879545,0.00004860712,0.00001081218,0.00000659827,0.00002850435,0.000002811771,0.9789934,0.0004074776,0.01792058,0.002540049,0.000008104084],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002396493,0.00027947,0.9925527,0.0002314865,0.00006946839,0.00004600873,0.00004579488,0.001246486,0.003132057],"genre_scores_gemma":[0.4307889,0.0009411195,0.5544857,0.0005578401,0.0002340577,0.0004287326,0.000394202,0.000486864,0.01168265],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006458944,"threshold_uncertainty_score":0.02160728,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0209271306621518,"score_gpt":0.2568127963650557,"score_spread":0.2358856657029039,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}