{"id":"W4386044647","doi":"10.48550/arxiv.2308.08977","title":"Hitting the High-Dimensional Notes: An ODE for SGD learning dynamics on GLMs and multi-index models","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Markov Chains and Monte Carlo Methods","field":"Mathematics","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; Natural Sciences and Engineering Research Council of Canada","keywords":"Mathematics; Applied mathematics; Ordinary differential equation; Stability (learning theory); Equivalence (formal languages); Rate of convergence; Covariance; Computer science; Statistics; Differential equation; Mathematical analysis; Machine learning; Discrete mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001288931,0.00039775,0.00047618,0.0001889342,0.0004462775,0.00007581233,0.0004658567,0.0003995171,0.000003214065],"category_scores_gemma":[0.0006267602,0.0003535114,0.0002296119,0.0001678939,0.0001164811,0.0001112965,0.0008205648,0.0009243655,3.374629e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002177962,"about_ca_system_score_gemma":0.00008738176,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003466667,"about_ca_topic_score_gemma":0.0006708549,"domain_scores_codex":[0.9979088,0.0003515166,0.000268187,0.000949754,0.0001244208,0.0003973148],"domain_scores_gemma":[0.996299,0.002280684,0.0003453371,0.0007319967,0.0001913661,0.0001515652],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001524006,0.0001042104,0.0005297851,0.0002690599,0.0001472018,0.00005187662,0.0005396619,0.7768313,0.00002306946,0.2195573,0.0000524123,0.001741647],"study_design_scores_gemma":[0.0007316581,0.00008560505,0.00006367324,0.0001910946,0.0001509368,0.000001881544,0.0006866718,0.88756,0.00002109323,0.110132,0.0000231841,0.0003521855],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.443045,0.00001067612,0.5556958,0.0001070336,0.000279059,0.0004984816,0.00005787042,0.0001755912,0.0001305195],"genre_scores_gemma":[0.9677489,0.00005628498,0.02781639,0.00008144999,0.0001281976,0.000006561632,0.00007539907,0.0001005081,0.003986315],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5278794,"threshold_uncertainty_score":0.9998917,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3284815676521052,"score_gpt":0.2874754231619534,"score_spread":0.04100614449015183,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}