{"id":"W4401662241","doi":"10.48550/arxiv.2407.11722","title":"Exploring Quantization for Efficient Pre-Training of Transformer Language Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies; Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Transformer; Computer science; Quantization (signal processing); Language model; Artificial intelligence; Engineering; Algorithm; Electrical engineering; Voltage","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002104035,0.001155414,0.0007470609,0.0004494788,0.0004575455,0.001416613,0.001855984,0.001099135,0.004484279],"category_scores_gemma":[0.0145773,0.0007385218,0.0005864045,0.0004629494,0.0009555329,0.00296801,0.002288813,0.003587901,0.001717159],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008927721,"about_ca_system_score_gemma":0.001859338,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004840416,"about_ca_topic_score_gemma":0.008667803,"domain_scores_codex":[0.9991645,0.0002964873,0.00006772641,0.0002207658,0.0001361846,0.0001144014],"domain_scores_gemma":[0.9968446,0.002146372,0.000130935,0.0004274665,0.000345275,0.000105257],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005795837,0.0002957478,0.00280039,0.0004094134,0.0001065729,0.0002202014,0.0005948357,0.5128546,0.03494882,0.02542537,0.007717052,0.4140474],"study_design_scores_gemma":[0.00003432359,0.00006448949,0.0001791493,0.00003025212,0.0000146669,0.0000361674,0.00004792699,0.9750775,0.01021288,0.01323048,0.001058281,0.0000138692],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04081139,0.0006066905,0.9505255,0.0005189205,0.00009325647,0.0001011474,0.0001784831,0.005223745,0.001940874],"genre_scores_gemma":[0.6597669,0.0003637545,0.3346941,0.0005856269,0.00004679767,0.0002597694,0.0006051309,0.001041584,0.002636291],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004840416,"threshold_uncertainty_score":0.01500136,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1879526093765473,"score_gpt":0.2415493246024011,"score_spread":0.05359671522585377,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}