{"id":"W4385462630","doi":"10.1016/j.artint.2023.103989","title":"Learning reward machines: A study in partially observable reinforcement learning","year":2023,"lang":"en","type":"article","venue":"Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":false,"ca_institutions":"Toronto Metropolitan University; Vector Institute; University of Toronto","funders":"Fondo Nacional de Desarrollo Científico y Tecnológico; Agencia Nacional de Investigación y Desarrollo; Government of Ontario; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; Microsoft Research","keywords":"Reinforcement learning; Observable; Computer science; Set (abstract data type); Artificial intelligence; Task (project management); Function (biology); Optimization problem; Representation (politics); Decomposition; Learning automata; Mathematical optimization; Automaton; Machine learning; Mathematics; Algorithm","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004299257,0.001016908,0.001656234,0.0007797962,0.000695892,0.002494436,0.002071107,0.002408694,0.002331899],"category_scores_gemma":[0.03181319,0.0008302978,0.0009455231,0.001373834,0.003859629,0.004742741,0.001270543,0.003792225,0.0001858075],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002024444,"about_ca_system_score_gemma":0.001718345,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005157377,"about_ca_topic_score_gemma":0.002325818,"domain_scores_codex":[0.9976562,0.001407465,0.00008953333,0.0002974244,0.0003752867,0.0001739236],"domain_scores_gemma":[0.9643662,0.03242844,0.001096032,0.000626743,0.001079592,0.0004029912],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008375951,0.0001089274,0.001036214,0.0002612844,0.00009352482,0.00007858474,0.0002438153,0.3641364,0.0004969015,0.6049277,0.0009198082,0.02761309],"study_design_scores_gemma":[0.00002600903,0.00005024113,0.0001732583,0.00002321245,0.00001514597,0.00001882148,0.00001984338,0.7905179,0.0001766199,0.2082769,0.0006907945,0.00001124081],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04996615,0.005683934,0.9318457,0.002379631,0.0001197093,0.00006414346,0.00005469507,0.0001500558,0.009736048],"genre_scores_gemma":[0.8927537,0.004374453,0.09637969,0.000308531,0.0003580289,0.0001726021,0.00007372163,0.00009619333,0.005483064],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005157377,"threshold_uncertainty_score":0.02273691,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08641297792112246,"score_gpt":0.3260962182458153,"score_spread":0.2396832403246928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}