{"id":"W3089723243","doi":"","title":"Learning Intrinsic Rewards as a Bi-Level Optimization Problem","year":2020,"lang":"en","type":"article","venue":"Uncertainty in Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Intrinsic motivation; Mathematical optimization; Mathematics; Psychology; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003017681,0.0009159143,0.002488755,0.0007877587,0.0005763099,0.003176725,0.002646972,0.004620814,0.006954757],"category_scores_gemma":[0.007924285,0.001120555,0.0009335757,0.001117602,0.002049184,0.003477801,0.003594024,0.003670876,0.0008517234],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002264591,"about_ca_system_score_gemma":0.001886153,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002897034,"about_ca_topic_score_gemma":0.00262548,"domain_scores_codex":[0.9984745,0.0005902721,0.00006842547,0.0003062119,0.0003546251,0.0002059469],"domain_scores_gemma":[0.9959875,0.0029453,0.0002623317,0.0001900053,0.000333149,0.0002817982],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008281993,0.00007457232,0.0003800557,0.000106733,0.00006118634,0.00006493591,0.00006034744,0.8974233,0.0005939419,0.08479225,0.001334747,0.0150251],"study_design_scores_gemma":[0.00001248088,0.00001771422,0.00004535299,0.000007193823,0.000005565926,0.000005525615,0.000004455778,0.9801939,0.00007962537,0.01940156,0.0002222576,0.000004391628],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02136815,0.000311094,0.9715674,0.001010004,0.00005280093,0.00005524568,0.00009724905,0.0001831777,0.005354892],"genre_scores_gemma":[0.7245764,0.0004193947,0.256562,0.000497216,0.0001665414,0.0005575037,0.0002569318,0.0002182181,0.01674581],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006954757,"threshold_uncertainty_score":0.02326602,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06927971827994504,"score_gpt":0.2923596390382754,"score_spread":0.2230799207583304,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}