{"id":"W2964855005","doi":"10.1609/aiide.v15i1.5220","title":"On Hard Exploration for Reinforcement Learning: A Case Study in Pommerman","year":2019,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence and Interactive Digital Entertainment","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; SAFER; Benchmark (surveying); Pruning; Computer science; Domain (mathematical analysis); Artificial intelligence; Machine learning; Computer security; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003639561,0.0007804218,0.001100206,0.0005978565,0.001359255,0.001190739,0.001438384,0.002866841,0.00372086],"category_scores_gemma":[0.01656688,0.0003516456,0.0009028999,0.0007155116,0.002848967,0.002455322,0.002190181,0.002921545,0.0002150864],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001301986,"about_ca_system_score_gemma":0.001285095,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003704427,"about_ca_topic_score_gemma":0.005262489,"domain_scores_codex":[0.9983986,0.000950055,0.00004632422,0.0001616277,0.0002436278,0.0001998146],"domain_scores_gemma":[0.9862834,0.01200132,0.0003632618,0.0005714763,0.0002612445,0.0005192549],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000309365,0.0002980797,0.002227052,0.0002393509,0.00006950188,0.0007869217,0.0004967743,0.8515396,0.0007924518,0.1178557,0.002456905,0.02292828],"study_design_scores_gemma":[0.00012838,0.0001503232,0.0004562929,0.00003494964,0.00001339282,0.0001275685,0.0001642038,0.8622957,0.0008350469,0.1331408,0.002634656,0.00001858787],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5021616,0.002515132,0.451159,0.006194767,0.0001348865,0.000348254,0.0004827948,0.0007762083,0.03622738],"genre_scores_gemma":[0.9294494,0.0003756485,0.06640702,0.0002634985,0.00003929119,0.0001403365,0.0001281005,0.00009220842,0.003104527],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00372086,"threshold_uncertainty_score":0.01924807,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1009064224702263,"score_gpt":0.3290656216999939,"score_spread":0.2281591992297676,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}