{"id":"W4306167947","doi":"10.1109/tpami.2022.3213503","title":"Robust Losses for Learning Value Functions","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Pattern Analysis and Machine Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; University of Alberta","funders":"","keywords":"Reinforcement learning; Mean squared error; Outlier; Clipping (morphology); Computer science; Bellman equation; Variance (accounting); Mathematical optimization; Robust statistics; Function (biology); Robust control; Sensitivity (control systems); Robust regression; Robustness (evolution); Mathematics; Artificial intelligence; Statistics; Control system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006731005,0.002079632,0.001562329,0.001210349,0.0005070758,0.002582038,0.001867489,0.002339866,0.004457855],"category_scores_gemma":[0.0285521,0.0007366391,0.001007563,0.0009044162,0.002844591,0.004106416,0.00274958,0.004014049,0.001111363],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002411715,"about_ca_system_score_gemma":0.001593241,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001223784,"about_ca_topic_score_gemma":0.0008304897,"domain_scores_codex":[0.9966718,0.001412298,0.0002095419,0.0005752255,0.0009104792,0.0002205953],"domain_scores_gemma":[0.9920185,0.005837456,0.0006305622,0.0006243644,0.0007171315,0.0001719906],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001008163,0.00005192557,0.0003181707,0.0001960421,0.00006339645,0.00007245945,0.00008459217,0.6395953,0.001559252,0.3133695,0.002652169,0.04193636],"study_design_scores_gemma":[0.00001517834,0.00004642759,0.00006151049,0.00004036805,0.000008435812,0.00002231526,0.000009255953,0.8509784,0.0007116797,0.1467491,0.001345165,0.00001219671],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.002016031,0.0003137127,0.9958315,0.0002188988,0.00002980263,0.00003319382,0.00003669053,0.0001357829,0.001384363],"genre_scores_gemma":[0.4854111,0.002254686,0.4988077,0.0007061682,0.0003350566,0.0008342696,0.0004539563,0.0007290782,0.01046811],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006731005,"threshold_uncertainty_score":0.03559738,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03400760162769441,"score_gpt":0.263753185754338,"score_spread":0.2297455841266436,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}