{"id":"W2971055146","doi":"","title":"Fast Convergence of Natural Gradient Descent for Over-Parameterized Neural Networks","year":2019,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Stochastic Gradient Optimization Techniques","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Initialization; Jacobian matrix and determinant; Gradient descent; Maxima and minima; Parameterized complexity; Convergence (economics); Artificial neural network; Applied mathematics; Mathematics; Computer science; Mathematical optimization; Algorithm; Mathematical analysis; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003564199,0.001005062,0.0009879468,0.0007636375,0.0006050431,0.0008523124,0.001039065,0.001214012,0.001630113],"category_scores_gemma":[0.01458867,0.0006169163,0.0006070875,0.0004297605,0.001539452,0.002012163,0.001732404,0.001779792,0.0002858369],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001526564,"about_ca_system_score_gemma":0.001273321,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003645505,"about_ca_topic_score_gemma":0.005394147,"domain_scores_codex":[0.9990693,0.000468167,0.00004418371,0.0001364005,0.0001922629,0.00008976889],"domain_scores_gemma":[0.9958447,0.002774525,0.0003336585,0.000412721,0.0005141593,0.0001202987],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001078765,0.00003264699,0.001184042,0.0001010995,0.00006542766,0.0001002477,0.00009596254,0.937412,0.002987081,0.03242952,0.0009560287,0.02452808],"study_design_scores_gemma":[0.000002212253,0.00000861596,0.00006485752,0.00000388501,0.000001402591,0.000008502184,0.000003130894,0.9946041,0.0002131758,0.004985891,0.0001022368,0.000002045621],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.037966,0.0003407456,0.9589761,0.0002283254,0.00002536133,0.00004620066,0.00003147293,0.0003707815,0.002015039],"genre_scores_gemma":[0.7549716,0.0003597178,0.2395097,0.0001869718,0.00004274494,0.0002148809,0.0001991123,0.0003314721,0.004183707],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003645505,"threshold_uncertainty_score":0.01884949,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03222558805984906,"score_gpt":0.1798108982464483,"score_spread":0.1475853101865993,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}