{"id":"W6910249080","doi":"10.48448/4a80-vz17","title":"LICHEE: Improving Language Model Pre-training with Multi-grained Tokenization","year":2021,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Language model; Variety (cybernetics); Natural language understanding; Natural language; Benchmark (surveying); Representation (politics); Language identification; Inference","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0009816658,0.0005906614,0.0005193339,0.001163988,0.0003183908,0.000392961,0.00106619,0.0002928661,0.0005931015],"category_scores_gemma":[0.0005535464,0.0005111447,0.00006408804,0.00268486,0.001029205,0.0003865703,0.0003738125,0.0004979947,0.0001483332],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003991658,"about_ca_system_score_gemma":0.002758645,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007888251,"about_ca_topic_score_gemma":0.004930721,"domain_scores_codex":[0.9957533,0.00006899412,0.0003943611,0.001414741,0.001387682,0.0009808801],"domain_scores_gemma":[0.9976721,0.00004165974,0.0005697173,0.001057783,0.0003474543,0.0003112344],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001237751,0.001148619,0.0009956496,0.0009994399,0.0003384883,0.0003866261,0.03784359,0.1095734,0.74243,0.003664817,0.02847126,0.07402438],"study_design_scores_gemma":[0.001012374,0.00006750358,0.00006734573,0.0005174372,0.00007384666,0.00003373243,0.001939824,0.9926224,0.001143543,0.00001500453,0.001682285,0.0008246913],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.006777081,0.001846048,0.7102472,0.000191807,0.0007116906,0.002571363,0.0005599303,0.003865469,0.2732294],"genre_scores_gemma":[0.1683024,0.00001302271,0.4140657,0.0003011613,0.0005125488,0.00006634137,0.000744187,0.001722633,0.4142721],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.883049,"threshold_uncertainty_score":0.999734,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03528229445213348,"score_gpt":0.3160258786409297,"score_spread":0.2807435841887962,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}