{"id":"W2750933313","doi":"","title":"Distributed Second-Order Optimization using Kronecker-Factored Approximations","year":2017,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Stochastic Gradient Optimization Techniques","field":"Computer Science","cited_by":72,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Stochastic gradient descent; Computer science; Computation; Overhead (engineering); Artificial neural network; Curvature; Algorithm; Machine learning; Scaling; Artificial intelligence; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001259782,0.001491032,0.001401315,0.0006953948,0.0005292895,0.001368831,0.001834884,0.00155109,0.005548259],"category_scores_gemma":[0.005070476,0.0006754549,0.001215432,0.0007404721,0.001243215,0.001595276,0.001402653,0.002310744,0.002409898],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001570702,"about_ca_system_score_gemma":0.00264406,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01526298,"about_ca_topic_score_gemma":0.02361878,"domain_scores_codex":[0.9993919,0.0001374782,0.00003593198,0.0001387838,0.0002225652,0.00007327985],"domain_scores_gemma":[0.9983338,0.0007625262,0.0001133301,0.0003137825,0.0003663312,0.0001103515],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005330813,0.00003626476,0.000501837,0.00006600987,0.00003314402,0.0000606196,0.00005154507,0.939988,0.001199962,0.02535835,0.003794902,0.028856],"study_design_scores_gemma":[0.000003713293,0.000003125243,0.00001637926,0.000002217198,9.474384e-7,0.000003656665,0.000001810042,0.9960103,0.0001104759,0.003513877,0.000331305,0.000002132304],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004528487,0.000157352,0.9917511,0.0001775053,0.00007034942,0.00002767845,0.00008620641,0.0009390658,0.002262292],"genre_scores_gemma":[0.2591709,0.0003385366,0.7262094,0.0003129062,0.000142697,0.0002435339,0.0006843308,0.00121307,0.01168463],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01526298,"threshold_uncertainty_score":0.0303483,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08007162583441403,"score_gpt":0.3591140077257221,"score_spread":0.279042381891308,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}