{"id":"W4382866902","doi":"10.1609/icaps.v33i1.27241","title":"Safe MDP Planning by Learning Temporal Patterns of Undesirable Trajectories and Averting Negative Side Effects","year":2023,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Automated Planning and Scheduling","topic":"Risk and Safety Analysis","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"National Research Foundation Singapore; National Research Foundation","keywords":"Computer science; Scalability; Lagrange multiplier; Reinforcement learning; Computation; Categorical variable; Fidelity; Markov process; Function (biology); Mathematical optimization; Action (physics); Artificial intelligence; Machine learning; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001480901,0.001058365,0.0008875373,0.0008549645,0.0004768607,0.000643032,0.001437269,0.0008027557,0.002554321],"category_scores_gemma":[0.006627616,0.0008162098,0.0009066499,0.0006029684,0.001164147,0.001728199,0.001237056,0.001943777,0.0002598986],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001525,"about_ca_system_score_gemma":0.00414006,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01028496,"about_ca_topic_score_gemma":0.01738134,"domain_scores_codex":[0.9994239,0.0001715288,0.00002888633,0.0001614699,0.0001470039,0.00006732157],"domain_scores_gemma":[0.9956598,0.003071686,0.0005179391,0.000235334,0.0003372247,0.0001780003],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004374889,0.00004503709,0.001142086,0.00004227455,0.00001999008,0.00003240424,0.0000340664,0.9737379,0.0003014539,0.005244331,0.0005073147,0.01884935],"study_design_scores_gemma":[0.000005689044,0.00001086947,0.00005538969,0.000003743479,0.000003298389,0.00000384806,0.000004724153,0.9926949,0.000177727,0.006925871,0.0001119618,0.000002000223],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05514339,0.0001318155,0.9413401,0.0004820825,0.0000198815,0.0001434165,0.0002262645,0.0007528131,0.00176034],"genre_scores_gemma":[0.7164451,0.0001439075,0.2795116,0.0001825663,0.00002399712,0.0003312205,0.0006416843,0.0001652339,0.002554807],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01028496,"threshold_uncertainty_score":0.02045017,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07393665330678277,"score_gpt":0.3543457930860298,"score_spread":0.280409139779247,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}