{"id":"W1037351197","doi":"10.1613/jair.4676","title":"Approximate Value Iteration with Temporally Extended Actions","year":2015,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Landmark; Computer science; Convergence (economics); Bellman equation; Mathematical optimization; Reinforcement learning; Value (mathematics); Function (biology); Term (time); State space; Mathematics; Artificial intelligence; Machine learning; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00181944,0.0007013217,0.0009005973,0.0004314267,0.0003636199,0.0008234325,0.001210394,0.001073595,0.00218793],"category_scores_gemma":[0.009021895,0.0004651643,0.00068642,0.0005604375,0.001545109,0.001880801,0.001528931,0.00147158,0.0002591783],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008666422,"about_ca_system_score_gemma":0.001103593,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002651111,"about_ca_topic_score_gemma":0.00295563,"domain_scores_codex":[0.9990249,0.0004388251,0.00006057854,0.0001635262,0.0002209964,0.00009116196],"domain_scores_gemma":[0.9963998,0.002617716,0.0003030895,0.0003036343,0.0002569566,0.0001188202],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001933607,0.00003391119,0.0009379204,0.00005970962,0.00003402476,0.00007814893,0.0001523689,0.9063586,0.001500281,0.04725663,0.0003501524,0.04304481],"study_design_scores_gemma":[0.00001406497,0.00003853927,0.00005273847,0.000008295465,0.000004173721,0.00001520862,0.0000110135,0.981024,0.0005599939,0.017954,0.0003122049,0.000005774678],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02330738,0.0001096339,0.9752405,0.00007882988,0.00001590351,0.00003171539,0.00002396324,0.0001831043,0.001008951],"genre_scores_gemma":[0.6523781,0.0001289923,0.3452871,0.00008075738,0.00001840452,0.0001931799,0.00009320434,0.00007835437,0.001741914],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002651111,"threshold_uncertainty_score":0.009622276,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3129026095091614,"score_gpt":0.4403874809011682,"score_spread":0.1274848713920068,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}