{"id":"W4382866902","doi":"10.1609/icaps.v33i1.27241","title":"Safe MDP Planning by Learning Temporal Patterns of Undesirable Trajectories and Averting Negative Side Effects","year":2023,"lang":"en","type":"article","venue":"Proceedings of the International Conference on Automated Planning and Scheduling","topic":"Risk and Safety Analysis","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"National Research Foundation Singapore; National Research Foundation","keywords":"Computer science; Scalability; Lagrange multiplier; Reinforcement learning; Computation; Categorical variable; Fidelity; Markov process; Function (biology); Mathematical optimization; Action (physics); Artificial intelligence; Machine learning; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001571986,0.0001761658,0.0003683093,0.0003378268,0.0003020101,0.0002750299,0.0004998741,0.00008937627,0.00001247245],"category_scores_gemma":[0.003304902,0.0001246038,0.00007848736,0.0004670636,0.0001080992,0.0003170968,0.0002007563,0.000311997,0.0000038157],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002120994,"about_ca_system_score_gemma":0.00003765719,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001217755,"about_ca_topic_score_gemma":0.000001899453,"domain_scores_codex":[0.9978564,0.00004641474,0.00058615,0.0003737321,0.0009249802,0.0002122755],"domain_scores_gemma":[0.9974301,0.00128812,0.0006882336,0.00007246288,0.000455121,0.00006599648],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002156093,0.00003067379,0.8788599,0.0001118911,0.0002402788,0.000005273623,0.009232528,0.01723867,0.08463778,0.004462735,0.0003247588,0.004639876],"study_design_scores_gemma":[0.0005440161,0.0001411138,0.1043318,0.001511861,0.00004325585,0.000008435421,0.01832111,0.8209628,0.04375404,0.01009344,0.00005257651,0.0002354717],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9957802,0.00007973429,0.0002401461,0.0007985521,0.0001365271,0.00008722059,0.00001516008,0.0001332635,0.002729245],"genre_scores_gemma":[0.9987041,0.00004819572,0.0006867548,0.00003669706,0.00002521477,0.00000487964,0.000006803336,0.00001092101,0.000476442],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8037242,"threshold_uncertainty_score":0.5081198,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07393665330678277,"score_gpt":0.3543457930860298,"score_spread":0.280409139779247,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}