{"id":"W3101629288","doi":"","title":"Delta-STN: Efficient Bilevel Optimization for Neural Networks using Structured Response Jacobians","year":2020,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Hyperparameter; Computer science; Mathematical optimization; Bilevel optimization; Artificial neural network; Jacobian matrix and determinant; Perceptron; Artificial intelligence; Machine learning; Optimization problem; Algorithm; Mathematics; Applied mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001601944,0.001334916,0.001119557,0.0005763233,0.0004821991,0.001163079,0.001662103,0.001704463,0.003796546],"category_scores_gemma":[0.005615646,0.0007929783,0.0006171609,0.0005904495,0.001264275,0.001779142,0.001949758,0.002391208,0.001190869],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001056704,"about_ca_system_score_gemma":0.001809104,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004603238,"about_ca_topic_score_gemma":0.007348876,"domain_scores_codex":[0.9995378,0.0001811981,0.00002988743,0.00008662462,0.0001179016,0.00004664178],"domain_scores_gemma":[0.998831,0.0006862422,0.00008430609,0.0001248884,0.0002108272,0.00006266639],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004030966,0.00002653088,0.0003004135,0.00006030469,0.00002458723,0.00003631464,0.00004490992,0.9463097,0.001221449,0.01339849,0.00155023,0.03698674],"study_design_scores_gemma":[0.000002841971,0.00000708745,0.00001241296,0.000004531242,0.000001139049,0.000003857943,0.000002672115,0.995104,0.0002129909,0.004428896,0.0002177922,0.000001896562],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.008830762,0.0002027442,0.9875726,0.0002023032,0.00003548009,0.00003860358,0.00005362296,0.0007349199,0.002329091],"genre_scores_gemma":[0.5019168,0.0003999645,0.4866007,0.0004228122,0.0000724622,0.0004601963,0.0004398453,0.000846659,0.008840594],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004603238,"threshold_uncertainty_score":0.01270074,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08668726734114977,"score_gpt":0.2088544038859706,"score_spread":0.1221671365448208,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}