{"id":"W3199151886","doi":"","title":"On Bonus-Based Exploration Methods in the Arcade Learning Environment","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Pace; Computer science; Suite; Set (abstract data type); Reinforcement learning; Domain (mathematical analysis); Hyperparameter; Artificial intelligence; Machine learning; Operations research; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.002851872,0.0006286629,0.0005506204,0.0004989446,0.000527764,0.0005455303,0.003135331,0.0004793475,0.0003374458],"category_scores_gemma":[0.0004229911,0.0006475613,0.0004264272,0.001337323,0.0004216896,0.0008368898,0.001381151,0.00249599,0.0003466985],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007229391,"about_ca_system_score_gemma":0.0003671975,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000266092,"about_ca_topic_score_gemma":0.00008864889,"domain_scores_codex":[0.9914328,0.004567889,0.000644534,0.00227815,0.0003580393,0.0007185986],"domain_scores_gemma":[0.9948223,0.002178603,0.0006081316,0.002115144,0.0001056715,0.0001701525],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006179707,0.0003649775,0.0006864053,0.00003056022,0.00003336765,0.000543387,0.005640216,0.9191115,0.0002806599,0.05595521,0.00001267807,0.01727921],"study_design_scores_gemma":[0.0001907818,0.000282337,0.0005353733,0.0002502616,0.00006298019,0.000002812742,0.005290571,0.9514299,0.007581777,0.03251687,0.001203879,0.0006524815],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1421357,0.0000903888,0.8544723,0.0007993149,0.0005847586,0.0005897682,0.000001448135,0.00007062161,0.001255737],"genre_scores_gemma":[0.9833058,0.0005438384,0.0148362,0.0004942117,0.00008547022,0.00000847651,0.00001693356,0.00003088933,0.0006782082],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8411701,"threshold_uncertainty_score":0.9998053,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2048932795810042,"score_gpt":0.2707503069292788,"score_spread":0.06585702734827462,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}