{"id":"W3196800830","doi":"10.1007/s10915-021-01628-3","title":"Stochastic Gradient Descent with Polyak’s Learning Rate","year":2021,"lang":"en","type":"article","venue":"Journal of Scientific Computing","topic":"Stochastic Gradient Optimization Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"Air Force Office of Scientific Research; Fundação para a Ciência e a Tecnologia; Institut de Valorisation des Données","keywords":"Stochastic gradient descent; Subgradient method; Mathematics; Constant (computer programming); Rate of convergence; Generalization; Gradient descent; Regular polygon; Descent (aeronautics); Applied mathematics; Convex function; Mathematical optimization; Convergence (economics); Artificial neural network; Mathematical analysis; Computer science; Artificial intelligence; Geometry","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004457125,0.001404714,0.002531295,0.0009726255,0.0009689097,0.002096703,0.002681911,0.004016095,0.005822189],"category_scores_gemma":[0.01775575,0.001357208,0.001206758,0.00152592,0.001843152,0.003186196,0.002520799,0.004929794,0.003126704],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001185536,"about_ca_system_score_gemma":0.002860553,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003825305,"about_ca_topic_score_gemma":0.003119466,"domain_scores_codex":[0.9978283,0.001065576,0.0001422545,0.0003105087,0.0005164719,0.0001368599],"domain_scores_gemma":[0.9936028,0.003856198,0.0002998501,0.0007070451,0.001277598,0.000256415],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001884997,0.0001840335,0.0003536572,0.0002129706,0.0001234368,0.00008196337,0.00004965113,0.7631475,0.002028821,0.1365839,0.009800442,0.08724492],"study_design_scores_gemma":[0.00001367146,0.00001537088,0.00002262577,0.000006784127,0.000005085482,0.000009709346,0.000001290511,0.9903327,0.0002377647,0.008746474,0.0006020517,0.000006556791],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002635799,0.0003001251,0.9943995,0.0003566178,0.0002571082,0.00003256825,0.0000276438,0.0003037663,0.001686979],"genre_scores_gemma":[0.2044246,0.0008636338,0.7715771,0.0006329553,0.0006290096,0.00048606,0.0003018343,0.0008549441,0.02022994],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005822189,"threshold_uncertainty_score":0.02357185,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01466382442135339,"score_gpt":0.2337888838771633,"score_spread":0.21912505945581,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}