{"id":"W4405784969","doi":"10.1109/iros58592.2024.10802547","title":"Meta SAC-Lag: Towards Deployable Safe Reinforcement Learning via MetaGradient-based Hyperparameter Tuning","year":2024,"lang":"en","type":"article","venue":"","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Victoria","funders":"Mitacs","keywords":"Hyperparameter; Reinforcement learning; Lag; Computer science; Meta learning (computer science); Reinforcement; Artificial intelligence; Machine learning; Engineering; Operating system; Structural engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001841879,0.001569941,0.001164561,0.0005066763,0.0003579123,0.001021196,0.001877893,0.001288419,0.002151552],"category_scores_gemma":[0.005384649,0.0007405988,0.0006732788,0.000278028,0.001324057,0.001194195,0.001864013,0.002748467,0.0008295153],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007433839,"about_ca_system_score_gemma":0.002038283,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003305048,"about_ca_topic_score_gemma":0.003979366,"domain_scores_codex":[0.9992526,0.000251606,0.00004179383,0.0001586882,0.000194164,0.0001010821],"domain_scores_gemma":[0.998426,0.0007496843,0.0002069949,0.0001965152,0.0002827485,0.0001381144],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009648978,0.00006555141,0.0008763463,0.00008858868,0.00005680348,0.00009669576,0.0001037552,0.937804,0.003436327,0.005895316,0.001565931,0.04991411],"study_design_scores_gemma":[0.00001075973,0.00002387526,0.00003592335,0.000005742732,0.000004239796,0.000008900355,0.000004246059,0.9977253,0.0004132976,0.001509495,0.0002546391,0.000003574909],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01130616,0.0001959686,0.984865,0.0001453262,0.00004573944,0.00004933083,0.0000257549,0.001825164,0.001541603],"genre_scores_gemma":[0.7657343,0.000198511,0.2298522,0.0004325016,0.00005771757,0.0002729031,0.0001750043,0.0004774009,0.002799649],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003305048,"threshold_uncertainty_score":0.009740889,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03878557997960261,"score_gpt":0.2639766936641335,"score_spread":0.2251911136845309,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}