{"id":"W3199151886","doi":"","title":"On Bonus-Based Exploration Methods in the Arcade Learning Environment","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Pace; Computer science; Suite; Set (abstract data type); Reinforcement learning; Domain (mathematical analysis); Hyperparameter; Artificial intelligence; Machine learning; Operations research; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008242547,0.002324811,0.001728208,0.001780934,0.0009830339,0.001777385,0.003060622,0.002534799,0.004831198],"category_scores_gemma":[0.02260445,0.000835035,0.001129136,0.000974761,0.002223688,0.004095099,0.003769134,0.005314826,0.001399968],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001647164,"about_ca_system_score_gemma":0.00243537,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006078199,"about_ca_topic_score_gemma":0.009698923,"domain_scores_codex":[0.9966461,0.00198914,0.0001238736,0.0004167279,0.0005573155,0.0002667386],"domain_scores_gemma":[0.9889852,0.008120019,0.0004616224,0.001167125,0.0007643286,0.0005017802],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001352801,0.0005004305,0.005606747,0.0004824702,0.0002560229,0.00009513547,0.0003219521,0.7092276,0.00174793,0.03153209,0.01201706,0.2368597],"study_design_scores_gemma":[0.0001179763,0.0002398605,0.0003303404,0.00008426385,0.00002535458,0.00005083687,0.00004766445,0.9766188,0.00082621,0.01889557,0.002737482,0.00002564697],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1629342,0.009423585,0.7873333,0.003840341,0.0006142957,0.0004554752,0.0004836108,0.01152522,0.02338991],"genre_scores_gemma":[0.6320138,0.001373645,0.3554948,0.001581347,0.0001843635,0.000536999,0.0008463127,0.001525422,0.006443383],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008242547,"threshold_uncertainty_score":0.04359126,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2048932795810042,"score_gpt":0.2707503069292788,"score_spread":0.06585702734827462,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}