{"id":"W7034215471","doi":"","title":"Towards accurate low-bitwidth BERT","year":2023,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Education, Innovation and Language Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Quantization (signal processing); Encoder; Pruning; Computational complexity theory; Edge device; Focus (optics); Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.002239793,0.0006585252,0.0006898637,0.0006058514,0.003999086,0.000258308,0.0009517505,0.0008327982,0.004720372],"category_scores_gemma":[0.003647182,0.0007044201,0.0003399025,0.002393204,0.0002003267,0.0007878454,0.0001143634,0.001229749,0.0009690645],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008677862,"about_ca_system_score_gemma":0.0003907685,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.01349015,"about_ca_topic_score_gemma":0.1344742,"domain_scores_codex":[0.9950514,0.000556402,0.0009536646,0.001022165,0.001423357,0.0009930236],"domain_scores_gemma":[0.9966614,0.0003718837,0.0006914058,0.0006170396,0.00132413,0.0003341351],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001130269,0.000345532,0.0001263212,0.0005210023,0.0005680181,0.00006676272,0.003409801,0.000005802625,0.001714955,0.6163941,0.001458169,0.3752765],"study_design_scores_gemma":[0.0006949367,0.00007016009,0.01260257,0.0006046906,0.0002073892,0.00000223185,0.0864174,0.000001903025,0.004557002,0.02721712,0.8659279,0.001696692],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5224048,0.0002356307,5.996676e-8,0.0002891501,0.00604483,0.0005996712,0.0003774588,0.0006914582,0.469357],"genre_scores_gemma":[0.7653481,0.001420664,0.0002242794,0.0009640139,0.0005919594,0.000350276,0.001867413,0.0002313229,0.2290019],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8644697,"threshold_uncertainty_score":0.9998088,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03855977822781113,"score_gpt":0.3408740432404509,"score_spread":0.3023142650126397,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}