{"id":"W4386473849","doi":"10.1007/978-3-031-43264-4_6","title":"Exploiting Reward Machines with Deep Reinforcement Learning in Continuous Action Domains","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University","funders":"York University","keywords":"Counterfactual thinking; Reinforcement learning; Computer science; Task (project management); Artificial intelligence; Action (physics); Machine learning","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009155257,0.0007405702,0.001025969,0.0003216008,0.0002553188,0.0009049239,0.001308828,0.001085159,0.003159392],"category_scores_gemma":[0.002806462,0.0005421276,0.0004298483,0.0004379638,0.0009273403,0.001356729,0.001396014,0.001971093,0.0004883003],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007328867,"about_ca_system_score_gemma":0.0006347023,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002396017,"about_ca_topic_score_gemma":0.002726761,"domain_scores_codex":[0.9996951,0.0001221595,0.00001417877,0.00006143571,0.0000621781,0.00004501399],"domain_scores_gemma":[0.9986746,0.0009770527,0.00009475651,0.0001053773,0.0000889048,0.00005927048],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000783004,0.00005913644,0.0003526716,0.00007553603,0.00003709038,0.00004695209,0.00002881074,0.8754249,0.001546038,0.03495056,0.00166706,0.08573295],"study_design_scores_gemma":[0.000003687579,0.000009214962,0.0000214363,0.000003247838,0.000001732021,0.000004477718,0.000001202598,0.9874893,0.0001689766,0.0121432,0.0001515091,0.000001999126],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01562911,0.0005232247,0.9802185,0.00025044,0.00006719021,0.0000178017,0.00004003755,0.0006124904,0.002641175],"genre_scores_gemma":[0.8030329,0.000491132,0.1904687,0.0001401862,0.00008473906,0.0001018925,0.000120445,0.0001528995,0.005407115],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003159392,"threshold_uncertainty_score":0.01056927,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02282712653940537,"score_gpt":0.2547727152913582,"score_spread":0.2319455887519528,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}