{"id":"W2865216059","doi":"10.24963/ijcai.2018/787","title":"Revisiting the Arcade Learning Environment: Evaluation Protocols and Open Problems for General Agents (Extended Abstract)","year":2018,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Variety (cybernetics); Key (lock); Field (mathematics); Focus (optics); Data science; Human–computer interaction; Artificial intelligence; Management science; Computer security; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1102686,0.001194679,0.001758741,0.002181817,0.002229661,0.009888913,0.00484834,0.003776159,0.01006877],"category_scores_gemma":[0.2646483,0.001319982,0.001419899,0.001581632,0.007846815,0.01823957,0.0126838,0.007836442,0.001986897],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005433915,"about_ca_system_score_gemma":0.007510677,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005085829,"about_ca_topic_score_gemma":0.004956671,"domain_scores_codex":[0.867625,0.105475,0.00572113,0.005513747,0.01372745,0.00193773],"domain_scores_gemma":[0.7344986,0.2044138,0.006851619,0.02270918,0.02699432,0.004532545],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001147419,0.0006114301,0.0024676,0.0006312194,0.0002066676,0.0001978582,0.003520165,0.07667667,0.001260135,0.7101402,0.01566905,0.1874717],"study_design_scores_gemma":[0.0005027899,0.0005561191,0.000506577,0.0004980379,0.00008065524,0.0001354269,0.001088916,0.4433229,0.004731331,0.5103078,0.0381275,0.0001418924],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01797732,0.0002601143,0.9606495,0.005170904,0.000162351,0.0008788041,0.0002196033,0.002093306,0.01258802],"genre_scores_gemma":[0.2695914,0.0002318919,0.7198464,0.001255604,0.0001049848,0.002068605,0.0004093458,0.0008929393,0.00559878],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8897314,"threshold_uncertainty_score":0.583163,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.157403217872697,"score_gpt":0.4028998136404876,"score_spread":0.2454965957677906,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}