{"id":"W2912943342","doi":"","title":"Improving generalization in reinforcement learning on Atari 2600 games","year":2019,"lang":"en","type":"article","venue":"International journal of advance research, ideas and innovations in technology","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Hyperparameter; Machine learning; Regularization (linguistics); Overfitting; Deep learning; Transfer of learning; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00126985,0.00009373749,0.0001645595,0.003383748,0.00005502845,0.0001092982,0.0008991689,0.00009446643,0.00001477765],"category_scores_gemma":[0.00148411,0.00009072343,0.00001825744,0.001672405,0.00009318613,0.0008268675,0.0003700095,0.0009717518,0.000009471064],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003602396,"about_ca_system_score_gemma":0.0001301659,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002966022,"about_ca_topic_score_gemma":0.000009543767,"domain_scores_codex":[0.998095,0.00006855091,0.0006561436,0.0002112484,0.0007030121,0.0002660122],"domain_scores_gemma":[0.9979885,0.000154897,0.0003554756,0.0002140056,0.001262106,0.00002503587],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002200743,0.00002524208,0.02547074,0.000007132342,0.00001165755,0.00002524225,0.00009844267,0.5789657,0.002690149,0.3503879,0.00002826348,0.04226755],"study_design_scores_gemma":[0.002665175,0.001590121,0.01296243,0.0006625226,0.000001814666,0.00013357,0.0006036264,0.9057221,0.004495669,0.04986744,0.0209802,0.0003152834],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3185229,0.0001762184,0.6739059,0.005368202,0.000481996,0.0002004121,2.4463e-7,0.0000316072,0.001312481],"genre_scores_gemma":[0.9798411,0.0004491177,0.01916998,0.0001421924,0.00004205012,0.000007832794,0.000002910599,0.000008053077,0.0003368309],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6613181,"threshold_uncertainty_score":0.4221832,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01771163940650869,"score_gpt":0.3392445821240863,"score_spread":0.3215329427175776,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}