{"id":"W2597452529","doi":"10.48550/arxiv.1703.04782","title":"Online Learning Rate Adaptation with Hypergradient Descent","year":2017,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Stochastic Gradient Optimization Techniques","field":"Computer Science","cited_by":77,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Stochastic gradient descent; Gradient descent; Rate of convergence; Range (aeronautics); Computation; Convergence (economics); Adaptation (eye); Online machine learning; Descent (aeronautics); Mode (computer interface); Artificial intelligence; Mathematical optimization; Machine learning; Algorithm; Active learning (machine learning); Mathematics; Artificial neural network; Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003400802,0.002052066,0.002043532,0.001099707,0.0005334177,0.00165777,0.003248314,0.002394,0.004623402],"category_scores_gemma":[0.01446787,0.001031673,0.0009807439,0.0009955737,0.001589738,0.002464507,0.002894844,0.003974655,0.003906806],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008427134,"about_ca_system_score_gemma":0.001720617,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0016182,"about_ca_topic_score_gemma":0.001744533,"domain_scores_codex":[0.9976811,0.0008972067,0.0001443833,0.0003774527,0.0007582783,0.0001414212],"domain_scores_gemma":[0.9971619,0.001075033,0.0003065288,0.0007415597,0.0005831976,0.0001317719],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001956591,0.0001650297,0.0008390514,0.0003178433,0.0001911624,0.0002350155,0.0001759429,0.6382617,0.01193788,0.08374368,0.01347756,0.2504595],"study_design_scores_gemma":[0.00002175232,0.00003164817,0.0000745441,0.00001730999,0.000007425911,0.00004660653,0.000004147759,0.98198,0.002433177,0.0121508,0.003218342,0.0000142836],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001119152,0.0001661383,0.996478,0.00009803872,0.00007548844,0.00003712612,0.00001810089,0.0008371997,0.001170857],"genre_scores_gemma":[0.1219119,0.0004922214,0.8674846,0.000405998,0.0003143827,0.0005152091,0.0001981915,0.001219989,0.007457551],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004623402,"threshold_uncertainty_score":0.0179854,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08496107352918601,"score_gpt":0.1946122829403172,"score_spread":0.1096512094111312,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}