{"id":"W2804948070","doi":"","title":"Using Reward Machines for High-Level Task Specification and Decomposition in Reinforcement Learning","year":2018,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":129,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Task (project management); Decomposition; Artificial intelligence; Human–computer interaction; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002548145,0.001008833,0.00138035,0.0004851958,0.0005417572,0.001418746,0.001537703,0.001567223,0.00428763],"category_scores_gemma":[0.009726472,0.0008914248,0.0008940281,0.000450998,0.001341304,0.002187766,0.002053584,0.003628187,0.0008703198],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001063457,"about_ca_system_score_gemma":0.001747118,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002853697,"about_ca_topic_score_gemma":0.004632657,"domain_scores_codex":[0.9986793,0.0005568902,0.00009434766,0.0002207119,0.0002696562,0.0001791415],"domain_scores_gemma":[0.995841,0.002909778,0.0002672012,0.0004107089,0.0003769085,0.0001944149],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000208343,0.0001549061,0.0006406398,0.0001236745,0.00004840447,0.00008591618,0.0001444381,0.8396086,0.005107183,0.03582956,0.001282078,0.1167662],"study_design_scores_gemma":[0.000008479065,0.00001578575,0.00003292048,0.000005340651,0.000003193726,0.000004263457,0.000003306382,0.987394,0.00072068,0.01164228,0.0001653805,0.000004351424],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004756275,0.00004125893,0.9940925,0.00006205506,0.00001619092,0.00003077303,0.00001706711,0.0005299415,0.0004539775],"genre_scores_gemma":[0.5462818,0.00009231898,0.4505089,0.0001492713,0.0000331814,0.0003952037,0.0001223788,0.0002844098,0.002132419],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00428763,"threshold_uncertainty_score":0.01434356,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09333430035172953,"score_gpt":0.3548435333796292,"score_spread":0.2615092330278997,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}