{"id":"W3185541071","doi":"","title":"Contextual Policy Transfer in Reinforcement Learning Domains via Deep Mixtures-of-Experts","year":2021,"lang":"en","type":"article","venue":"Uncertainty in Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Robustness (evolution); Machine learning; Artificial intelligence; Transfer of learning; Reuse; Context (archaeology); Task (project management); Bayesian probability; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00352456,0.001447765,0.001741789,0.0006059097,0.0004881209,0.001026695,0.002157267,0.001858052,0.002147847],"category_scores_gemma":[0.01088927,0.0008842503,0.0008115894,0.0005226043,0.001693613,0.002167935,0.002774528,0.003175586,0.0004400747],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001701917,"about_ca_system_score_gemma":0.001600064,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007612758,"about_ca_topic_score_gemma":0.006177522,"domain_scores_codex":[0.9986381,0.0005989851,0.00006409166,0.000296518,0.0002271971,0.0001751829],"domain_scores_gemma":[0.9960001,0.002854253,0.0002896045,0.0003019958,0.000317488,0.0002366002],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001535168,0.00009070586,0.0005535864,0.00004942336,0.00004284756,0.00004710373,0.0001004657,0.9580431,0.0007348411,0.006849368,0.0005253716,0.03280957],"study_design_scores_gemma":[0.00001191632,0.00002408943,0.00004034377,0.000004832873,0.000004040332,0.000004863537,0.0000046085,0.9940261,0.0002988031,0.005446339,0.0001294931,0.000004597972],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04928366,0.0005257522,0.9467129,0.0003251756,0.00005135998,0.00008450114,0.00005910325,0.001131472,0.001826148],"genre_scores_gemma":[0.8996539,0.0001831456,0.09770652,0.0002376732,0.00003973391,0.0001477966,0.000114701,0.0001124933,0.001804061],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007612758,"threshold_uncertainty_score":0.01863986,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03181191723208893,"score_gpt":0.2972237819968313,"score_spread":0.2654118647647424,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}