{"id":"W3130843035","doi":"","title":"Conservative Safety Critics for Exploration","year":2021,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Convergence (economics); Task (project management); Suite; Upper and lower bounds; Artificial intelligence; Machine learning; Mathematical optimization; Engineering; Mathematics; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003427426,0.001523781,0.001264877,0.0008666893,0.0007102448,0.001498278,0.0018139,0.001676376,0.003260151],"category_scores_gemma":[0.02140095,0.0007500566,0.0007847692,0.0003562761,0.003185916,0.002227191,0.003132825,0.004427667,0.0006799951],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001290597,"about_ca_system_score_gemma":0.002000398,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001510068,"about_ca_topic_score_gemma":0.001651699,"domain_scores_codex":[0.9981446,0.0006519448,0.00008949143,0.0003346818,0.0005892685,0.000190067],"domain_scores_gemma":[0.9866975,0.009744322,0.001006267,0.001162457,0.0009308944,0.0004585662],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001447386,0.00004372089,0.001080569,0.0001141378,0.00003647535,0.0001221385,0.0001487666,0.8891531,0.002065313,0.08092644,0.001588472,0.02457607],"study_design_scores_gemma":[0.00001976943,0.00004340584,0.00007399048,0.00002237949,0.000007249618,0.00003090813,0.00001062101,0.9561567,0.0007106579,0.04228693,0.0006281428,0.000009310877],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01274006,0.000181563,0.9832035,0.0003787286,0.00004586706,0.00004418065,0.00004437142,0.0005418358,0.002819842],"genre_scores_gemma":[0.8612038,0.0002989043,0.1312736,0.0004854281,0.0001128172,0.0003865905,0.0001891605,0.0002779422,0.005771711],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003427426,"threshold_uncertainty_score":0.01812619,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.117497022015405,"score_gpt":0.379141317792146,"score_spread":0.261644295776741,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}