{"id":"W3034995656","doi":"10.24963/ijcai.2020/452","title":"Closing the Generalization Gap of Adaptive Gradient Methods in Training Deep Neural Networks","year":2020,"lang":"en","type":"article","venue":"","topic":"Stochastic Gradient Optimization Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Canadian Institute for Advanced Research; University of Pennsylvania","keywords":"Stochastic gradient descent; Gradient descent; Computer science; Artificial neural network; Convergence (economics); Generalization; Closing (real estate); Artificial intelligence; Momentum (technical analysis); Gradient method; Stationary point; Deep learning; Rate of convergence; Algorithm; Mathematical optimization; Mathematics; Key (lock); Law","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006920116,0.001294033,0.001182148,0.0006463628,0.0005888526,0.001142022,0.001412679,0.001619502,0.001382891],"category_scores_gemma":[0.02508292,0.0007236548,0.0007865783,0.0006888628,0.002269288,0.002838581,0.00250797,0.003852788,0.0004873198],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000856728,"about_ca_system_score_gemma":0.001520839,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002623524,"about_ca_topic_score_gemma":0.002313192,"domain_scores_codex":[0.9974449,0.001333565,0.0001835211,0.0003630792,0.0005530829,0.0001220583],"domain_scores_gemma":[0.991197,0.006072931,0.0005017111,0.001186129,0.0008596667,0.0001826745],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002640089,0.00007745416,0.002324225,0.0002419643,0.0001475545,0.0001370681,0.0002927728,0.7378103,0.004540056,0.07096955,0.004170612,0.1790245],"study_design_scores_gemma":[0.00001077936,0.00004614367,0.0001480009,0.00002303366,0.000007405546,0.0000287305,0.000009808697,0.9809826,0.001209411,0.01662083,0.0009067272,0.000006407206],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02040304,0.001089124,0.9752467,0.0008535575,0.0001049015,0.00003720525,0.00002455545,0.0006114608,0.001629466],"genre_scores_gemma":[0.5474535,0.001550766,0.4455309,0.000843711,0.0002333811,0.0002406524,0.0001942957,0.0006731842,0.003279637],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006920116,"threshold_uncertainty_score":0.03659749,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1055841967632464,"score_gpt":0.3277355163136001,"score_spread":0.2221513195503537,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}