{"id":"W3035618017","doi":"","title":"Improving Transformer Optimization Through Better Initialization","year":2020,"lang":"en","type":"article","venue":"International Conference on Machine Learning","topic":"Power Transformer Diagnostics and Insulation","field":"Engineering","cited_by":51,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Initialization; Computer science; Transformer; Electrical engineering; Engineering; Voltage; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008492885,0.001249458,0.0009560338,0.0005390692,0.0003832105,0.001092156,0.0006967647,0.001250147,0.0053432],"category_scores_gemma":[0.003604975,0.0007183032,0.0006563325,0.0004572845,0.0003897307,0.001646485,0.0008777447,0.001715759,0.001545867],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004966898,"about_ca_system_score_gemma":0.0008146251,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00407753,"about_ca_topic_score_gemma":0.005500712,"domain_scores_codex":[0.9996178,0.0001533705,0.00002571092,0.00008416147,0.00007592732,0.00004303527],"domain_scores_gemma":[0.9993043,0.0003165984,0.00005050707,0.0001560232,0.0001433088,0.00002926285],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002539342,0.0001348457,0.00129538,0.00007988269,0.00007912691,0.00007845904,0.000058326,0.7722101,0.0127412,0.004918465,0.00739134,0.2007589],"study_design_scores_gemma":[0.000009605436,0.00001178683,0.0001288324,0.000004104132,0.000005671918,0.00001168367,0.000003351116,0.9961598,0.00243297,0.0006958169,0.0005329829,0.000003405954],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02726988,0.0002728679,0.963923,0.0003304598,0.0001429722,0.00003621322,0.00009199604,0.003185693,0.004746927],"genre_scores_gemma":[0.5480723,0.0001948332,0.4416825,0.0003679732,0.00009463492,0.00009049853,0.0006922156,0.001220988,0.007584141],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0053432,"threshold_uncertainty_score":0.01787478,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02893562775711362,"score_gpt":0.2536630670916855,"score_spread":0.2247274393345718,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}