{"id":"W4392305758","doi":"10.5220/0012567700003654","title":"Mitigating Outlier Activations in Low-Precision Fine-Tuning of Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University; Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Outlier; Language model; Artificial intelligence; Natural language processing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002449068,0.001294504,0.001444714,0.00084386,0.0006855338,0.001801892,0.001880631,0.001513549,0.001731175],"category_scores_gemma":[0.017716,0.0006563197,0.0005074029,0.000836233,0.0006325186,0.001910685,0.001983423,0.003315872,0.0009953376],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000614034,"about_ca_system_score_gemma":0.001823515,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006160534,"about_ca_topic_score_gemma":0.01279074,"domain_scores_codex":[0.9982085,0.0004454619,0.0001107689,0.0005360788,0.0004752309,0.0002239142],"domain_scores_gemma":[0.9951121,0.002751997,0.0003603969,0.0009051049,0.0006815756,0.0001887532],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001786159,0.0006254482,0.01275254,0.0003488711,0.0003845877,0.0005439712,0.0004376127,0.2113352,0.08311164,0.004191001,0.009079949,0.675403],"study_design_scores_gemma":[0.00004029699,0.0001481711,0.002779463,0.00002322638,0.0000595585,0.0001683585,0.00007506581,0.9691144,0.01953307,0.006539918,0.001497698,0.0000207406],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1758783,0.001567429,0.8090194,0.000782742,0.0004108634,0.00008637174,0.000318513,0.01007873,0.001857567],"genre_scores_gemma":[0.8680943,0.0002911238,0.1277071,0.0004291527,0.0001248228,0.00007645795,0.0006243236,0.000653766,0.001998879],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006160534,"threshold_uncertainty_score":0.01295203,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01691056866944777,"score_gpt":0.2983856961191145,"score_spread":0.2814751274496667,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}