{"id":"W2150468603","doi":"10.1613/jair.3912","title":"The Arcade Learning Environment: An Evaluation Platform for General Agents","year":2013,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1060,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alberta Innovates; University of Alberta; Compute Canada","keywords":"Testbed; Benchmarking; Reinforcement learning; Benchmark (surveying); Imitation; Interface (matter)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01337618,0.001908025,0.001078831,0.002286951,0.0005852241,0.001953681,0.003830818,0.001737645,0.008489219],"category_scores_gemma":[0.03481341,0.0007657617,0.0009738894,0.001327463,0.001255557,0.002862258,0.003909424,0.003113078,0.002936229],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001480088,"about_ca_system_score_gemma":0.002318647,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004746819,"about_ca_topic_score_gemma":0.004241815,"domain_scores_codex":[0.9887922,0.007331612,0.0008097087,0.0008219923,0.001816693,0.00042776],"domain_scores_gemma":[0.9787911,0.01330225,0.0009883412,0.002991008,0.002880726,0.00104653],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003549396,0.003639912,0.0098472,0.003319166,0.000914829,0.000463411,0.001117868,0.4374845,0.007879542,0.05099973,0.1444713,0.3363132],"study_design_scores_gemma":[0.001494366,0.002020892,0.002326968,0.0002504004,0.0001246794,0.0001865743,0.0002384754,0.8897074,0.00915134,0.02375511,0.07060668,0.0001371563],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07411522,0.00135767,0.7953433,0.001674257,0.0007121144,0.004908601,0.008251042,0.08102206,0.03261568],"genre_scores_gemma":[0.2756532,0.000527038,0.6941121,0.0006190911,0.00009082857,0.007379126,0.009865311,0.006918795,0.004834517],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01337618,"threshold_uncertainty_score":0.07074082,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2902597031161736,"score_gpt":0.4427010381262976,"score_spread":0.152441335010124,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}