{"id":"W4382239700","doi":"10.1609/aaai.v37i6.25825","title":"PiCor: Multi-Task Deep Reinforcement Learning with Policy Correction","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Task (project management); Computer science; Constraint (computer-aided design); Set (abstract data type); Artificial intelligence; Interference (communication); Machine learning; Channel (broadcasting); Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001926429,0.001219045,0.001346895,0.0003439855,0.000389271,0.0007683884,0.003089683,0.001424501,0.0027492],"category_scores_gemma":[0.003847777,0.0006650815,0.0005673752,0.0003307752,0.0009797805,0.001317276,0.001932713,0.00272927,0.0007552951],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009635744,"about_ca_system_score_gemma":0.002451269,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005203695,"about_ca_topic_score_gemma":0.006875366,"domain_scores_codex":[0.9992281,0.0001902363,0.00003897101,0.0001879379,0.0002272566,0.0001274467],"domain_scores_gemma":[0.9989567,0.0003905301,0.0001448873,0.0002259854,0.0001668737,0.0001150246],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000252871,0.0002924805,0.0008863619,0.0001430783,0.0001085695,0.0001071653,0.00005383503,0.8058909,0.004812209,0.010409,0.005087529,0.1719562],"study_design_scores_gemma":[0.00001828806,0.00003947597,0.00003980065,0.000003242904,0.000004104712,0.000009552774,0.000001512458,0.9974614,0.0007249803,0.001291563,0.0004012624,0.000004830844],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009938279,0.0002390283,0.9837285,0.0001935887,0.00007603156,0.0001165244,0.00005937573,0.003768253,0.001880475],"genre_scores_gemma":[0.6186112,0.0002051821,0.3733476,0.0006232777,0.00008507055,0.0004175097,0.0002990174,0.0004571083,0.005953927],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005203695,"threshold_uncertainty_score":0.01034683,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06260084640833517,"score_gpt":0.2989850664917558,"score_spread":0.2363842200834206,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}