{"id":"W3206540493","doi":"10.1109/icra48506.2021.9561333","title":"Shaping Rewards for Reinforcement Learning with Imperfect Demonstrations using Generative Models","year":2021,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Vector Institute; University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Generative grammar; Artificial intelligence; Machine learning; Generative model; Adversarial system; Function (biology); Action (physics); State space; Convergence (economics); Bellman equation; Range (aeronautics); Imperfect; Imitation; Mathematical optimization; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001723392,0.0009310507,0.001081016,0.0005586173,0.0003900459,0.0008173189,0.001473519,0.001165824,0.002667452],"category_scores_gemma":[0.007652349,0.0006802382,0.0006856419,0.0003586927,0.002041058,0.00153484,0.001936071,0.002293781,0.0003846229],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001316526,"about_ca_system_score_gemma":0.001178097,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003006422,"about_ca_topic_score_gemma":0.003540474,"domain_scores_codex":[0.9993414,0.0002639608,0.00002558461,0.0001307636,0.0001585854,0.00007969495],"domain_scores_gemma":[0.9962162,0.002934663,0.0002611128,0.0002562779,0.0001880587,0.0001435608],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003417653,0.00002283164,0.0003400068,0.00002777297,0.0000139992,0.00004733984,0.00004486203,0.9672207,0.0007591182,0.0178486,0.0003284373,0.01331215],"study_design_scores_gemma":[0.00000458071,0.00001101129,0.00002765906,0.000004174508,0.000002261763,0.00000807756,0.000002228877,0.9923152,0.0002369082,0.007242395,0.0001423819,0.000003149462],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01416796,0.0001287679,0.983597,0.0002026772,0.00001998839,0.00003212245,0.0000283484,0.0003346713,0.001488425],"genre_scores_gemma":[0.8842568,0.0001862821,0.1110104,0.0002007899,0.00003612815,0.0002146275,0.0001005836,0.0001642871,0.003829915],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003006422,"threshold_uncertainty_score":0.009552181,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07543834377601084,"score_gpt":0.2929367701417165,"score_spread":0.2174984263657057,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}