{"id":"W7093331850","doi":"10.1016/j.engappai.2025.112779","title":"Shaping <mml:math xmlns:mml=\"http://www.w3.org/1998/Math/MathML\" altimg=\"si159.svg\" display=\"inline\" id=\"d1e703\"> <mml:mi>Q</mml:mi> </mml:math> -values right: A distributional normalized actor–critic approach","year":2025,"lang":"en","type":"article","venue":"Engineering Applications of Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; National Fire Agency; Busan Metropolitan City; Information Technology Research Centre; Korea Institute for Advancement of Technology; Ministry of Science and ICT, South Korea; Ministry of Trade, Industry and Energy; Ministry of Education","keywords":"Reinforcement learning; Key (lock); Upper and lower bounds; Stability (learning theory); Range (aeronautics); Value (mathematics); Q-learning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001125764,0.0009125342,0.0006787375,0.0009026182,0.0005479123,0.003149503,0.002049932,0.001439105,0.2078201],"category_scores_gemma":[0.007277266,0.0006062378,0.0006010652,0.000884388,0.0009659234,0.003571461,0.002432074,0.002212659,0.1026817],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001203558,"about_ca_system_score_gemma":0.001148475,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004612441,"about_ca_topic_score_gemma":0.007676829,"domain_scores_codex":[0.9991605,0.0002263475,0.00006302947,0.0002062286,0.0002900088,0.00005390657],"domain_scores_gemma":[0.9983577,0.0004814229,0.00006851913,0.0004436131,0.0005485727,0.0001001154],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002517325,0.00009252918,0.0004631947,0.0003577312,0.00002996606,0.0001828136,0.0002391668,0.02307739,0.005704495,0.3413099,0.3738461,0.254445],"study_design_scores_gemma":[0.0000744386,0.00003394126,0.0004469452,0.0001073634,0.00001439038,0.0001768885,0.00007176496,0.2053446,0.01449654,0.24415,0.5350263,0.00005677802],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001302634,0.000130402,0.821421,0.002028473,0.0005409277,0.0001334756,0.006228208,0.02726245,0.1409525],"genre_scores_gemma":[0.1207911,0.0005370078,0.5402162,0.001671088,0.0003309271,0.0006025368,0.01455345,0.03315508,0.2881426],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2078201,"threshold_uncertainty_score":0.6952274,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0226693454117592,"score_gpt":0.2668609660288699,"score_spread":0.2441916206171107,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}