{"id":"W2891236810","doi":"","title":"Reinforcement Learning with Multiple Experts: A Bayesian Model Combination Approach","year":2018,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Convergence (economics); Artificial intelligence; Machine learning; Bayesian probability; Bellman equation; Domain (mathematical analysis); Function (biology); Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004810173,0.001604131,0.002525439,0.001059388,0.00061574,0.001457387,0.002973491,0.0026478,0.003276682],"category_scores_gemma":[0.01063564,0.001177797,0.001175676,0.0007915839,0.001745095,0.002508775,0.002545097,0.003238474,0.0006847056],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001387102,"about_ca_system_score_gemma":0.001441906,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002602858,"about_ca_topic_score_gemma":0.002880552,"domain_scores_codex":[0.9971352,0.001459032,0.00009654789,0.0003901598,0.0006950215,0.0002240652],"domain_scores_gemma":[0.9952195,0.003162924,0.0005304661,0.0003020804,0.0005295891,0.0002553524],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007446039,0.00005781961,0.0003847503,0.00005109706,0.00009488788,0.00007553061,0.00007284006,0.9492175,0.000645974,0.02335042,0.0006464614,0.02532834],"study_design_scores_gemma":[0.00001272354,0.00002440352,0.00004072569,0.000007131869,0.0000111794,0.00001502437,0.000003589209,0.9881025,0.0001529075,0.01135773,0.0002636933,0.000008388519],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007028116,0.0002055434,0.9899048,0.0003013879,0.00002684569,0.00004729076,0.00001671978,0.0001522016,0.002317084],"genre_scores_gemma":[0.7487891,0.0003916981,0.2444531,0.0004262807,0.0001569685,0.0003986055,0.00008267655,0.000104916,0.005196577],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004810173,"threshold_uncertainty_score":0.0254389,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01966117805250088,"score_gpt":0.2382829261831284,"score_spread":0.2186217481306275,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}