{"id":"W7132871321","doi":"","title":"Who Should I Trust? Uncertainty and Risk for Knowledge Transfer from Multiple Sources in Reinforcement Learning Domains","year":2023,"lang":"","type":"dissertation","venue":"TSpace","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Knowledge transfer; Transfer of learning; Set (abstract data type); Inference; Quality (philosophy); Bayesian inference; Uncertainty quantification; Bayesian probability","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001665093,0.001184798,0.001380329,0.0008010436,0.001083677,0.0008190325,0.001357821,0.000955447,0.0001044022],"category_scores_gemma":[0.001564217,0.001279454,0.0003638687,0.00118228,0.0002065805,0.0004729242,0.0003417569,0.002126966,0.0001163767],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004222266,"about_ca_system_score_gemma":0.0004329783,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006760195,"about_ca_topic_score_gemma":0.005167226,"domain_scores_codex":[0.9936468,0.0005323701,0.001500812,0.001917042,0.000933218,0.001469763],"domain_scores_gemma":[0.9940554,0.003610769,0.0006565011,0.0009555374,0.0003441871,0.0003775927],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003584407,0.00004351375,0.01202553,0.0004095847,0.0002576494,0.000008887934,0.1749859,0.8022833,0.000201065,0.0005087003,0.0001141425,0.008803304],"study_design_scores_gemma":[0.004124588,0.000852032,0.009699032,0.001225204,0.0002931097,9.205683e-7,0.02921095,0.9443755,0.0006774419,0.0001440725,0.008104589,0.001292515],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4654488,0.0008102744,0.5292394,0.0001218528,0.001287414,0.001959684,0.00001395242,0.0002639594,0.0008546709],"genre_scores_gemma":[0.9302728,0.004910568,0.003833532,0.00005470944,0.0004154029,0.0004787827,0.001065628,0.0002543443,0.05871428],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5254059,"threshold_uncertainty_score":0.9998538,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04253784223620904,"score_gpt":0.326914110571627,"score_spread":0.284376268335418,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}