{"id":"W2970803838","doi":"","title":"Lookahead Optimizer: k steps forward, 1 step back","year":2019,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":176,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Treebank; Computer science; Stochastic gradient descent; Hyperparameter; Artificial neural network; Machine translation; Computation; Artificial intelligence; Gradient descent; Algorithm; Deep neural networks; Stability (learning theory); Mathematical optimization; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00139436,0.002282244,0.002000096,0.0005904149,0.0005439692,0.001158388,0.002618356,0.002587322,0.009455135],"category_scores_gemma":[0.003832751,0.00105402,0.0008048912,0.0006425067,0.0009128733,0.001626361,0.001449244,0.003083411,0.006158217],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009125455,"about_ca_system_score_gemma":0.002323115,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006823099,"about_ca_topic_score_gemma":0.01049019,"domain_scores_codex":[0.9993363,0.0001780223,0.00005454863,0.000189898,0.0001724585,0.00006873939],"domain_scores_gemma":[0.999222,0.0003311397,0.0000695203,0.0001743044,0.0001478067,0.0000550663],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008049755,0.0003517779,0.001428415,0.0003365876,0.0002042052,0.0002265959,0.0001077944,0.613649,0.006171588,0.01376205,0.04105739,0.3218997],"study_design_scores_gemma":[0.00005539183,0.00004963327,0.0000752905,0.00001188822,0.00001140654,0.00002441736,0.000007253946,0.993634,0.001476932,0.003242313,0.001400877,0.00001061439],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01681593,0.001110017,0.9568142,0.0005395073,0.000297029,0.0001888526,0.0004042523,0.01915729,0.004672947],"genre_scores_gemma":[0.2647398,0.0004347577,0.7129027,0.001095183,0.0001556224,0.0005940909,0.002017586,0.002466964,0.0155933],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009455135,"threshold_uncertainty_score":0.03163064,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01359282323168757,"score_gpt":0.2465341774647059,"score_spread":0.2329413542330183,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}