{"id":"W4417239585","doi":"10.1038/s41467-025-66009-y","title":"Discovery of the reward function for embodied reinforcement learning agents","year":2025,"lang":"en","type":"article","venue":"Nature Communications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Victoria","funders":"National Key Research and Development Program of China; State Key Laboratory of Industrial Control Technology; State Key Laboratory of Mechanical Transmissions; Huazhong University of Science and Technology; National Natural Science Foundation of China","keywords":"Embodied cognition; Reinforcement learning; Regret; Maximization; Adaptability; Cognition; Function (biology); Cognitive robotics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003360526,0.0000919375,0.0001144168,0.00009565994,0.0004981177,0.00009270881,0.002706365,0.0001277619,0.000001390452],"category_scores_gemma":[0.0005102294,0.00007186548,0.0001208124,0.0005410482,0.00007971196,0.0002941197,0.001233863,0.0006257228,0.000002618265],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007570995,"about_ca_system_score_gemma":0.0001078227,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000006912497,"about_ca_topic_score_gemma":0.00001309445,"domain_scores_codex":[0.9991015,0.0001167126,0.0002874663,0.0001491957,0.0002013748,0.0001438183],"domain_scores_gemma":[0.9969082,0.0003764687,0.0002268227,0.00227498,0.0001994009,0.00001407188],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00000988184,0.00002581106,0.002691945,0.00003002406,0.00007324777,1.868942e-8,0.0002320337,0.2620584,0.0002885651,0.7266114,0.006713761,0.001264868],"study_design_scores_gemma":[0.0006430456,0.00009930681,0.0158051,0.0001755434,0.00008088817,5.169005e-7,0.0001113804,0.6962799,0.001123049,0.003167402,0.2823335,0.0001803829],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0002299758,0.000427831,0.9718778,0.004635181,0.000688808,0.0004633837,8.631596e-7,0.00007309895,0.02160304],"genre_scores_gemma":[0.9787047,0.00009513633,0.01233911,0.0006996098,0.00001227851,0.00005897859,0.00002166312,0.000005527112,0.008063007],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9784747,"threshold_uncertainty_score":0.5029145,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02517842891644697,"score_gpt":0.3064949102456547,"score_spread":0.2813164813292078,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}