{"id":"W2754879180","doi":"10.1613/jair.5699","title":"Revisiting the Arcade Learning Environment: Evaluation Protocols and Open Problems for General Agents","year":2018,"lang":"en","type":"preprint","venue":"Journal of Artificial Intelligence Research","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":46,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alberta Innovates; Alberta Innovates - Technology Futures; Alberta Machine Intelligence Institute; Compute Canada; DeepMind; National Science Foundation","keywords":"Variety (cybernetics); Computer science; Benchmark (surveying); Key (lock); Field (mathematics); Data science; Best practice; State (computer science); Management science; Artificial intelligence; Engineering; Computer security; Political science","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1092359,0.00157611,0.002031175,0.002430534,0.001975174,0.008701421,0.005152726,0.004062979,0.008125558],"category_scores_gemma":[0.3080233,0.001185015,0.001257238,0.002136852,0.006246079,0.01533696,0.01014672,0.009704608,0.002485888],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005616248,"about_ca_system_score_gemma":0.00810021,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005822661,"about_ca_topic_score_gemma":0.006563233,"domain_scores_codex":[0.8508615,0.1193537,0.005649713,0.006876897,0.01563162,0.001626518],"domain_scores_gemma":[0.7420573,0.1925093,0.006582883,0.02889262,0.02618293,0.003774914],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001562071,0.0007472308,0.00472513,0.001117001,0.0004364427,0.0001814562,0.002596735,0.1308234,0.001467112,0.5359082,0.03554575,0.2848896],"study_design_scores_gemma":[0.000585775,0.0006794519,0.0008370691,0.0006034026,0.0001091905,0.0001247693,0.0007289476,0.5693612,0.003951241,0.3831101,0.03976535,0.0001436379],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0226384,0.0009048047,0.9435588,0.009530804,0.000414702,0.0009384098,0.0006010251,0.004081619,0.01733155],"genre_scores_gemma":[0.2790711,0.0005345293,0.7079653,0.002315576,0.0002308871,0.002501197,0.0008710578,0.001540488,0.004969928],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8907641,"threshold_uncertainty_score":0.5777017,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4920271193673282,"score_gpt":0.5281472299857938,"score_spread":0.03612011061846554,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}