{"id":"W4382203111","doi":"10.1609/aaai.v37i8.26186","title":"Fast Convergence in Learning Two-Layer Neural Networks with Separable Data","year":2023,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Stochastic Gradient Optimization Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"National Science Foundation","keywords":"Overfitting; Generalization; Convergence (economics); Separable space; Applied mathematics; Stability (learning theory); Mathematics; Artificial neural network; Gradient descent; Stochastic gradient descent; Exponential function; Exponential stability; Computer science; Mathematical optimization; Algorithm; Artificial intelligence; Mathematical analysis; Nonlinear system; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004486981,0.00113418,0.0009017101,0.0006342708,0.0003525942,0.0007711748,0.001352008,0.001122642,0.001232102],"category_scores_gemma":[0.01876759,0.0006961265,0.0005860508,0.00057953,0.002036887,0.002850236,0.0023494,0.002187852,0.0002614689],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002041393,"about_ca_system_score_gemma":0.00137192,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006481566,"about_ca_topic_score_gemma":0.004694685,"domain_scores_codex":[0.9990211,0.0004004205,0.0000625157,0.0001881658,0.0002170134,0.0001108654],"domain_scores_gemma":[0.99484,0.00374233,0.0002886072,0.0004147716,0.000587346,0.0001269128],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001148672,0.00003115271,0.0008358142,0.0001018589,0.00004012486,0.00005606937,0.0001068817,0.9285754,0.001734684,0.0370047,0.0005247971,0.03087368],"study_design_scores_gemma":[0.000004479714,0.00001615114,0.00005032993,0.000004938935,0.000001930948,0.000006610853,0.000004094716,0.9893022,0.0004197492,0.01008881,0.00009857478,0.000002156365],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04574298,0.0004315004,0.951525,0.0003402696,0.00002897659,0.0000386292,0.00003992801,0.0004188043,0.001433843],"genre_scores_gemma":[0.7271969,0.0003662537,0.2675828,0.0002613103,0.00003400374,0.0002004,0.0002020862,0.0002122064,0.003944046],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006481566,"threshold_uncertainty_score":0.02372968,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1136395192751324,"score_gpt":0.3207995889363792,"score_spread":0.2071600696612468,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}