{"id":"W6947979547","doi":"10.48448/xpf1-k446","title":"Contextual Policy Transfer in Reinforcement Learning Domains via Deep Mixtures-of-Experts","year":2021,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Species Distribution and Climate Change","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Robustness (evolution); Reuse; Transfer of learning; Task (project management); Bayesian probability; Policy learning; Q-learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0003668413,0.000270485,0.0003358421,0.0003240695,0.0001300586,0.00004725876,0.0005648014,0.0001861364,0.1335113],"category_scores_gemma":[0.0001048835,0.000247859,0.00007196971,0.00128913,0.001293779,0.00009757933,0.0002540968,0.0002375512,0.000189902],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001017145,"about_ca_system_score_gemma":0.0001269349,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003529793,"about_ca_topic_score_gemma":0.008662975,"domain_scores_codex":[0.9975716,0.00005426626,0.0003722743,0.0005661977,0.0008550269,0.0005806363],"domain_scores_gemma":[0.9993232,0.00003086104,0.0001092026,0.0003402052,0.00001953795,0.0001770007],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000229266,0.003018709,0.02681077,0.0006425781,0.000237476,0.0005120324,0.0167269,0.01532511,0.5300454,0.07574353,0.201582,0.1291261],"study_design_scores_gemma":[0.002301023,0.000420689,0.003685937,0.0003339659,0.0000311046,0.00004166541,0.005470129,0.005811249,0.009249448,0.0001387694,0.9711511,0.001364883],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.005408753,0.0004320387,0.007722127,0.0004062552,0.0003040184,0.0004205406,0.00002311323,0.00008171776,0.9852014],"genre_scores_gemma":[0.8897519,0.0005448377,0.0003236444,0.0008020408,0.0001571022,0.00003447356,0.0002816236,0.0001273304,0.1079771],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.8843431,"threshold_uncertainty_score":0.9999974,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02042084115145876,"score_gpt":0.2806425444615363,"score_spread":0.2602217033100775,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}