{"id":"W4385574336","doi":"10.18653/v1/2022.findings-emnlp.363","title":"Improving Generalization of Pre-trained Language Models via Stochastic Weight Averaging","year":2022,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"","keywords":"Generalization; Computer science; Language model; Artificial intelligence; Flatness (cosmology); Computation; Distillation; Machine learning; Convergence (economics); Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002182199,0.00007625289,0.00009997979,0.0001004147,0.0001163099,0.00002804823,0.0004901342,0.00001520027,0.00007195382],"category_scores_gemma":[0.000008830488,0.00007639312,0.00003579987,0.0002115221,0.000007245814,0.0003125486,0.0004383841,0.00007830924,9.146354e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005392296,"about_ca_system_score_gemma":0.00004547095,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003156776,"about_ca_topic_score_gemma":0.000006879078,"domain_scores_codex":[0.9990412,0.0000497884,0.0002042671,0.0002572649,0.0002848187,0.0001626567],"domain_scores_gemma":[0.9994507,0.00002377725,0.00008561709,0.0003736972,0.00003170986,0.0000344652],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002233846,0.00001932434,0.000009966749,0.00001195461,0.000005536205,0.000002696661,0.003937211,0.9133669,0.01752143,0.04544528,0.00001734787,0.01966019],"study_design_scores_gemma":[0.0001653897,0.0000199304,0.0000123797,0.000002712883,0.000003138043,0.000007717527,0.00005890915,0.9941705,0.001693591,0.003771646,0.000005287427,0.00008873659],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04758227,0.00007006863,0.9513167,0.0001091653,0.0001915451,0.0001219094,0.000001283551,0.0001355033,0.0004716039],"genre_scores_gemma":[0.9040449,2.930969e-7,0.0952908,0.0001585739,0.00003617277,0.00001756684,0.000003628401,0.000007293085,0.0004407674],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8564627,"threshold_uncertainty_score":0.3115221,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0105345872943328,"score_gpt":0.2211012462258387,"score_spread":0.2105666589315059,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}