{"id":"W4400972116","doi":"10.2316/j.2024.206-1114","title":"OVERCOMING VALUE OVERESTIMATION FOR DISTRIBUTIONAL REINFORCEMENT LEARNING-BASED PATH PLANNING WITH CONSERVATIVE CONSTRAINTS","year":2024,"lang":"en","type":"article","venue":"International Journal of Robotics and Automation","topic":"Robotic Path Planning Algorithms","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Reinforcement; Path (computing); Value (mathematics); Computer science; Mathematical optimization; Operations research; Artificial intelligence; Psychology; Mathematics; Machine learning; Social psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000475437,0.0001098902,0.0001316644,0.0001848695,0.00008998268,0.0003942988,0.000217519,0.00004172919,0.000002895329],"category_scores_gemma":[0.00019744,0.00009075861,0.00005104144,0.0001016894,0.00005759764,0.0006177867,0.00003157845,0.0001579353,0.000001129538],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001471285,"about_ca_system_score_gemma":0.000273727,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000002275355,"about_ca_topic_score_gemma":4.661139e-8,"domain_scores_codex":[0.9988307,0.00003687124,0.0003679123,0.000143014,0.0005040467,0.0001174516],"domain_scores_gemma":[0.9986089,0.000474549,0.0003218427,0.00005440537,0.0004842216,0.00005605715],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002500945,0.00001646685,0.0007696472,0.00002743778,0.0001007088,0.0000562667,0.000264661,0.9356077,0.0001299721,0.05875069,0.0001648664,0.004086563],"study_design_scores_gemma":[0.0006717693,0.00025993,0.003430339,0.0008615172,0.00002244128,0.0002219299,0.00003127164,0.991825,0.0002422633,0.002070913,0.0002577523,0.0001049181],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004313749,0.00009496692,0.9926996,0.002044489,0.0006258136,0.0001070735,0.000009988014,0.00005435135,0.0000500247],"genre_scores_gemma":[0.7165943,0.000003350937,0.2831658,0.000079816,0.00009320315,0.000003288581,0.00004048735,0.000005524051,0.00001426399],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7122805,"threshold_uncertainty_score":0.3802233,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01897315524903784,"score_gpt":0.2928042851376677,"score_spread":0.2738311298886298,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}