{"id":"W3020277389","doi":"10.48550/arxiv.2004.12399","title":"Reinforcement Learning Generalization with Surprise Minimization","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Surprise; Generalization; Computer science; Reinforcement learning; Artificial intelligence; Robustness (evolution); Machine learning; Mathematics; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004033348,0.00141974,0.001450072,0.0003587299,0.0004881635,0.001055792,0.001880234,0.001771198,0.001936773],"category_scores_gemma":[0.01563885,0.0004538253,0.000692883,0.0002476247,0.002136803,0.001903219,0.002286287,0.00293383,0.0003678116],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001660608,"about_ca_system_score_gemma":0.00148491,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003105982,"about_ca_topic_score_gemma":0.002715954,"domain_scores_codex":[0.998285,0.000639739,0.00009996871,0.0004365301,0.0003334595,0.000205328],"domain_scores_gemma":[0.993646,0.003566982,0.0006619377,0.001174861,0.0005688991,0.000381311],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002350043,0.000198348,0.001897707,0.0001279702,0.0001093167,0.0001242161,0.00008630644,0.9375467,0.003890379,0.01494512,0.00198941,0.0388496],"study_design_scores_gemma":[0.00002776318,0.0001439305,0.0002577482,0.00001026717,0.000009746283,0.00002379861,0.00001091497,0.9790738,0.001152299,0.01895585,0.0003248357,0.000009141007],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2767963,0.0004762023,0.7084391,0.001828268,0.0001650478,0.000293749,0.0002294561,0.001934203,0.009837671],"genre_scores_gemma":[0.9528551,0.00008143541,0.04452199,0.0003282316,0.00003537924,0.0001842026,0.0001484655,0.0001151919,0.001730194],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004033348,"threshold_uncertainty_score":0.0213306,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06140952338521455,"score_gpt":0.1835702624956942,"score_spread":0.1221607391104796,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}