{"id":"W3211684788","doi":"","title":"Average-Reward Learning and Planning with Options","year":2021,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Markov decision process; Computer science; Abstraction; Convergence (economics); Sample complexity; Artificial intelligence; Machine learning; Mathematical proof; Markov chain; Sample (material); Domain (mathematical analysis); Markov process; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002110958,0.0007984567,0.0008541836,0.0005686482,0.0004563681,0.001201547,0.002029548,0.0009456483,0.004209871],"category_scores_gemma":[0.008165511,0.0004325076,0.0008047282,0.0008361548,0.001548346,0.003487159,0.0019436,0.002296028,0.0004515895],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001162374,"about_ca_system_score_gemma":0.001330165,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002590043,"about_ca_topic_score_gemma":0.002416388,"domain_scores_codex":[0.9990107,0.0004076159,0.00005556112,0.0001975669,0.0002306454,0.0000978502],"domain_scores_gemma":[0.9972995,0.001704883,0.0002521347,0.0003194037,0.000240554,0.0001835642],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009632625,0.00005819317,0.0005522778,0.000086195,0.00004901525,0.00008130753,0.0001168669,0.6938518,0.0009540542,0.240513,0.0009390777,0.06270196],"study_design_scores_gemma":[0.00001116522,0.000030967,0.00005793298,0.000009061611,0.000006032438,0.00001643116,0.000006883192,0.8864444,0.0004849579,0.1120992,0.000824876,0.000008191575],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009048671,0.0001558558,0.988221,0.0001458416,0.00002687183,0.00001938621,0.00003229286,0.0001688881,0.002181262],"genre_scores_gemma":[0.6124535,0.0004227194,0.3819631,0.000162077,0.00005837754,0.0001812759,0.0001439121,0.0001194098,0.00449563],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004209871,"threshold_uncertainty_score":0.01408345,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01646724830475379,"score_gpt":0.2526428685441239,"score_spread":0.2361756202393701,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}