{"id":"W3034724428","doi":"10.48550/arxiv.2003.00203","title":"Contextual Policy Transfer in Reinforcement Learning Domains via Deep Mixtures-of-Experts","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Reuse; Transfer of learning; Artificial intelligence; Robustness (evolution); Task (project management); Context (archaeology); Machine learning; Policy learning; Dynamics (music); Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00336829,0.001347436,0.001652448,0.000546434,0.00045936,0.001054935,0.002219943,0.001787965,0.002060666],"category_scores_gemma":[0.01120327,0.0009033659,0.0007621152,0.0005151306,0.001845231,0.002287689,0.002563445,0.003198143,0.0004372425],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00162033,"about_ca_system_score_gemma":0.001475103,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006858031,"about_ca_topic_score_gemma":0.005853498,"domain_scores_codex":[0.9986805,0.000625058,0.00005812609,0.0002864487,0.0001968433,0.0001530056],"domain_scores_gemma":[0.9963368,0.002627851,0.0002575704,0.0002906949,0.0002763793,0.0002106045],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001288319,0.0000750681,0.0005969953,0.00004646974,0.00004074751,0.00004620099,0.0001036911,0.9617606,0.000687609,0.009260635,0.0005342958,0.02671888],"study_design_scores_gemma":[0.00001124123,0.0000218707,0.00004156886,0.000004690473,0.000003987276,0.00000505908,0.000004695266,0.9922856,0.0002744982,0.007197294,0.0001452514,0.000004269617],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03869808,0.000418243,0.9580233,0.0003546355,0.00004463429,0.00006442569,0.00004846919,0.0008407869,0.001507423],"genre_scores_gemma":[0.9037486,0.0001978184,0.09321253,0.0002250332,0.00004488169,0.0001483744,0.0001052351,0.000131345,0.002186169],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006858031,"threshold_uncertainty_score":0.01781344,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05375636255390823,"score_gpt":0.2086563624918888,"score_spread":0.1548999999379806,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}