{"id":"W3176692111","doi":"10.18653/v1/2021.findings-acl.119","title":"LICHEE: Improving Language Model Pre-training with Multi-grained Tokenization","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Lexical analysis; Language model; Benchmark (surveying); Artificial intelligence; Natural language processing; Inference; Variety (cybernetics); Natural language understanding; Representation (politics); Natural language; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002204554,0.002294614,0.001878877,0.001900827,0.0008585251,0.001268566,0.00314673,0.0016158,0.005689275],"category_scores_gemma":[0.006913162,0.001073605,0.001581704,0.001751338,0.000817798,0.005321814,0.002931249,0.005041662,0.005197589],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009651424,"about_ca_system_score_gemma":0.002803972,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01296001,"about_ca_topic_score_gemma":0.0296892,"domain_scores_codex":[0.9985241,0.0004702417,0.0000970507,0.00049515,0.0002188901,0.0001946286],"domain_scores_gemma":[0.9969274,0.001660932,0.000112918,0.0006994924,0.000459747,0.0001394945],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000482215,0.0005163775,0.003012528,0.0003514534,0.0003024937,0.0003354878,0.0004191519,0.1639594,0.01945934,0.005935372,0.03781062,0.7674156],"study_design_scores_gemma":[0.00005546412,0.0001158413,0.0004919382,0.00002438743,0.00004774918,0.00009849724,0.00008926934,0.9772376,0.0102324,0.005148555,0.006410383,0.00004783605],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03098411,0.001412377,0.932345,0.0004486816,0.0003925011,0.000185976,0.001062665,0.03063524,0.002533534],"genre_scores_gemma":[0.3407575,0.0008979687,0.6233751,0.001386637,0.0002916917,0.0008209375,0.01388985,0.00340987,0.01517046],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01296001,"threshold_uncertainty_score":0.02576917,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03371922272302266,"score_gpt":0.2704040480036245,"score_spread":0.2366848252806018,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}