{"id":"W4405603119","doi":"10.2316/j.2025.206-1114","title":"OVERCOMING VALUE OVERESTIMATION FOR DISTRIBUTIONAL REINFORCEMENT LEARNING-BASED PATH PLANNING WITH CONSERVATIVE CONSTRAINTS, 124-132.","year":2024,"lang":"en","type":"article","venue":"International Journal of Robotics and Automation","topic":"Robotic Path Planning Algorithms","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Reinforcement; Path (computing); Computer science; Mathematical optimization; Motion planning; Value (mathematics); Artificial intelligence; Mathematics; Machine learning; Engineering; Structural engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003153642,0.0008188261,0.001163881,0.0005120402,0.0007903707,0.001362118,0.001887852,0.001107775,0.003171493],"category_scores_gemma":[0.02127592,0.0006939364,0.0004132143,0.0006190154,0.001683503,0.00305754,0.002799951,0.002874303,0.0003848992],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001615339,"about_ca_system_score_gemma":0.002768092,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005589304,"about_ca_topic_score_gemma":0.009852014,"domain_scores_codex":[0.9984036,0.0006528968,0.0001002245,0.0002578358,0.0004057827,0.0001796936],"domain_scores_gemma":[0.9920905,0.005901683,0.000369894,0.0005280235,0.0008515002,0.0002583688],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005747513,0.0002092576,0.002028847,0.0002238841,0.00007094703,0.0002018383,0.0002776857,0.771399,0.004383341,0.04898445,0.004579185,0.1670669],"study_design_scores_gemma":[0.00002305111,0.00005420207,0.0001642091,0.00001730012,0.00001259296,0.00003262657,0.00003399946,0.9648219,0.001263448,0.03302235,0.0005468588,0.000007396299],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04156803,0.0004707962,0.9516103,0.0007956037,0.00009771581,0.00007986673,0.00007276416,0.000389016,0.004915916],"genre_scores_gemma":[0.8303732,0.0002259018,0.1650854,0.0002597936,0.00005559973,0.0001491641,0.0001290638,0.0001779499,0.003543834],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005589304,"threshold_uncertainty_score":0.01667821,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01889957237805722,"score_gpt":0.29031087367882,"score_spread":0.2714113013007628,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}