{"id":"W3211684788","doi":"","title":"Average-Reward Learning and Planning with Options","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Markov decision process; Computer science; Abstraction; Convergence (economics); Sample complexity; Artificial intelligence; Machine learning; Mathematical proof; Markov chain; Sample (material); Domain (mathematical analysis); Markov process; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0002005589,0.0001266742,0.0001387106,0.0001141407,0.000420471,0.001728651,0.0001738616,0.00005184548,0.000002082192],"category_scores_gemma":[0.0000877347,0.000109051,0.00001659048,0.0004021172,0.0000287134,0.004035474,0.000107887,0.0002502815,0.00002099675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003423974,"about_ca_system_score_gemma":0.0001001751,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004959339,"about_ca_topic_score_gemma":1.586153e-7,"domain_scores_codex":[0.9988878,0.00006305236,0.0003362145,0.0001487628,0.0003473436,0.0002168713],"domain_scores_gemma":[0.9991324,0.00004566163,0.0002688273,0.0001504264,0.0003286919,0.00007401485],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002020981,0.000002186353,0.001993291,0.0001823206,0.000006575748,0.000008865442,0.002559854,0.988095,0.00004194686,0.001624611,0.00007769853,0.005405614],"study_design_scores_gemma":[0.0002491462,0.00005062263,0.0004164628,0.0002201397,0.000004219972,0.0002896089,0.000925723,0.9869106,0.0001129953,0.000006527532,0.01065636,0.0001575653],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007471422,0.0003627709,0.9867648,0.0002099795,0.0002688216,0.0001096271,2.875058e-7,0.0003311759,0.004481129],"genre_scores_gemma":[0.9908171,0.000009759126,0.008077201,0.0001629527,0.0000504558,0.00001477646,0.00001718702,0.000006944933,0.0008435905],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9833457,"threshold_uncertainty_score":0.9993076,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01646724830475379,"score_gpt":0.2526428685441239,"score_spread":0.2361756202393701,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}