{"id":"W4386473849","doi":"10.1007/978-3-031-43264-4_6","title":"Exploiting Reward Machines with Deep Reinforcement Learning in Continuous Action Domains","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University","funders":"York University","keywords":"Counterfactual thinking; Reinforcement learning; Computer science; Task (project management); Artificial intelligence; Action (physics); Machine learning","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0014305,0.0006804811,0.0006912414,0.001536827,0.0003925019,0.0007773743,0.002454758,0.0003119819,0.00001467628],"category_scores_gemma":[0.0002616865,0.0006110641,0.0001107364,0.001248927,0.0004268376,0.0009494318,0.00143075,0.00182182,0.00008156888],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006432075,"about_ca_system_score_gemma":0.0003877787,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007933777,"about_ca_topic_score_gemma":0.0002769021,"domain_scores_codex":[0.9948716,0.0000734806,0.000838088,0.001566872,0.001598723,0.001051275],"domain_scores_gemma":[0.9971562,0.0006937701,0.0006441565,0.001117127,0.0002237154,0.0001650304],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001092799,0.000005778495,0.0006427105,0.0000475105,0.00001196146,0.0001772359,0.001165867,0.8598932,0.00003892613,0.00401427,0.000002632435,0.133989],"study_design_scores_gemma":[0.0005281289,0.0004378599,0.0003010114,0.0009009914,0.000008691947,0.00006487283,0.000002880673,0.9914085,0.0001804481,0.005016367,0.0004027666,0.0007474914],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0002623951,0.00005035457,0.9927163,0.0003698578,0.001224428,0.0005568345,2.203982e-7,0.0005631367,0.004256434],"genre_scores_gemma":[0.6191608,0.0001976021,0.3716422,0.001104328,0.0007873236,0.00006393898,0.00003027405,0.0002033126,0.00681024],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.6210741,"threshold_uncertainty_score":0.9996341,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02282712653940537,"score_gpt":0.2547727152913582,"score_spread":0.2319455887519528,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}