{"id":"W4403534366","doi":"10.1109/codit62066.2024.10708505","title":"Combining Dense and Sparse Rewards to Improve Deep Reinforcement Learning Policies in Reach-Avoid Games with Faster Evaders in Two vs. One Scenarios","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Royal Military College of Canada; Queen's University","funders":"","keywords":"Reinforcement learning; Computer science; Reinforcement; Artificial intelligence; Human–computer interaction; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001809459,0.001272477,0.0009642591,0.0004381677,0.0003238882,0.0006623414,0.001022909,0.0009350058,0.001240324],"category_scores_gemma":[0.006618001,0.000401162,0.0003250012,0.0002136813,0.0009451222,0.00128553,0.001474708,0.001603736,0.0001969282],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009106064,"about_ca_system_score_gemma":0.001187661,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003326881,"about_ca_topic_score_gemma":0.003883333,"domain_scores_codex":[0.9994922,0.000169308,0.00002730156,0.00009520662,0.0001089363,0.000107067],"domain_scores_gemma":[0.9976214,0.001494065,0.0002681091,0.0001418695,0.0002342956,0.0002403638],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000114955,0.0001313014,0.001155632,0.00004274524,0.00002788658,0.00005244309,0.00004323905,0.9688956,0.001625334,0.004619559,0.0003227758,0.02296842],"study_design_scores_gemma":[0.0000129964,0.00007481564,0.0001039157,0.00000448587,0.000005528892,0.000008681659,0.000004972504,0.9971668,0.000351743,0.002156429,0.0001054408,0.000004210839],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2928787,0.0004963799,0.7003539,0.000566859,0.0001052775,0.0001085917,0.00005634637,0.0007194454,0.004714575],"genre_scores_gemma":[0.9722425,0.0000595317,0.0265134,0.00008779558,0.00001490098,0.00004263023,0.00002425537,0.00002783595,0.000987076],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003326881,"threshold_uncertainty_score":0.009569466,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01728873881557427,"score_gpt":0.2657106130266615,"score_spread":0.2484218742110873,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}