{"id":"W7093331850","doi":"10.1016/j.engappai.2025.112779","title":"Shaping <mml:math xmlns:mml=\"http://www.w3.org/1998/Math/MathML\" altimg=\"si159.svg\" display=\"inline\" id=\"d1e703\"> <mml:mi>Q</mml:mi> </mml:math> -values right: A distributional normalized actor–critic approach","year":2025,"lang":"en","type":"article","venue":"Engineering Applications of Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; National Fire Agency; Busan Metropolitan City; Information Technology Research Centre; Korea Institute for Advancement of Technology; Ministry of Science and ICT, South Korea; Ministry of Trade, Industry and Energy; Ministry of Education","keywords":"Reinforcement learning; Key (lock); Upper and lower bounds; Stability (learning theory); Range (aeronautics); Value (mathematics); Q-learning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0004658686,0.0002423803,0.0001818623,0.0001745315,0.0003733871,0.0002644783,0.001181685,0.0001963634,0.00001110876],"category_scores_gemma":[0.000317546,0.0003005964,0.0001899953,0.0007374006,0.000189706,0.0003771323,0.0004484144,0.0003840955,0.0001998127],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000037273,"about_ca_system_score_gemma":0.0001814017,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005795454,"about_ca_topic_score_gemma":0.000003562082,"domain_scores_codex":[0.9978044,0.00002747606,0.0007166002,0.0005052943,0.0005010931,0.0004451267],"domain_scores_gemma":[0.99827,0.0003645228,0.0002446477,0.0008731349,0.0001276906,0.0001200092],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007491493,0.00005789727,0.000002754822,0.000120865,0.00004782922,0.000002398413,0.0002020068,0.3401814,0.001229765,0.6570319,0.00002726381,0.001088446],"study_design_scores_gemma":[0.00004254013,0.00004906852,0.00002670311,0.000137777,0.00004515566,0.00001425561,0.0001007337,0.9730271,0.0235143,0.001509173,0.001306972,0.000226228],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1275769,0.0001195474,0.8710831,0.0001603977,0.0002829378,0.0001104533,0.00002373644,0.0002298079,0.0004130969],"genre_scores_gemma":[0.936542,0.00003836397,0.06262662,0.00006274888,0.0001645327,0.0003565106,0.0001367692,0.00003070545,0.00004179292],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.808965,"threshold_uncertainty_score":0.9999446,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0226693454117592,"score_gpt":0.2668609660288699,"score_spread":0.2441916206171107,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}