{"id":"W3176194860","doi":"","title":"SGD in the Large: Average-case Analysis, Asymptotics, and Stepsize Criticality","year":2021,"lang":"en","type":"article","venue":"Conference on Learning Theory","topic":"Statistical Mechanics and Entropy","field":"Physics and Astronomy","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Convergence (economics); Stochastic gradient descent; Applied mathematics; Mathematics; Criticality; Rate of convergence; Limit (mathematics); Statistical physics; Mathematical optimization; Computer science; Mathematical analysis; Physics; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003448596,0.0008762229,0.00104737,0.001477006,0.0007684032,0.00187325,0.001699476,0.001178409,0.003076253],"category_scores_gemma":[0.02688495,0.0005652046,0.0009547686,0.0007140791,0.003133572,0.004556122,0.002519175,0.002936778,0.000415959],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001857923,"about_ca_system_score_gemma":0.001156239,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002052843,"about_ca_topic_score_gemma":0.001665575,"domain_scores_codex":[0.9988205,0.0003885477,0.00005501478,0.0002402427,0.0003452382,0.0001504678],"domain_scores_gemma":[0.9875925,0.008959731,0.0009306864,0.001104321,0.0009091548,0.0005036161],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00006742102,0.00006640222,0.001578533,0.0001637288,0.00006878712,0.0002704462,0.0001883573,0.3530551,0.003946422,0.6253328,0.002134021,0.01312799],"study_design_scores_gemma":[0.000004587479,0.00001478687,0.000178903,0.00001401557,0.000005753756,0.00003628068,0.00001205016,0.8652208,0.0007282647,0.133254,0.0005191955,0.0000113659],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05393384,0.001071983,0.9358569,0.001202461,0.00007796931,0.000046843,0.00009183431,0.0004370124,0.007281103],"genre_scores_gemma":[0.865642,0.001242162,0.1268908,0.0004575547,0.0001985677,0.0002505286,0.000219785,0.0004419649,0.004656817],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003448596,"threshold_uncertainty_score":0.01823813,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01944880602390112,"score_gpt":0.2931511756311167,"score_spread":0.2737023696072156,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}