{"id":"W4387424400","doi":"10.1016/j.asoc.2023.110901","title":"BERT models for Brazilian Portuguese: Pretraining, evaluation and tokenization analysis","year":2023,"lang":"en","type":"article","venue":"Applied Soft Computing","topic":"Topic Modeling","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico","keywords":"Computer science; Artificial intelligence; Natural language processing; Language model; Transformer; Lexical analysis; Sentence; Encoder; Transfer of learning; Portuguese; Textual entailment; Machine translation; Logical consequence; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003522891,0.002237887,0.001161549,0.001482864,0.001357162,0.002069579,0.001919603,0.001335425,0.009205168],"category_scores_gemma":[0.01386055,0.0009409004,0.001330176,0.001188248,0.0004130099,0.003354956,0.001270878,0.003326569,0.005702895],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002369212,"about_ca_system_score_gemma":0.002993728,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.08541442,"about_ca_topic_score_gemma":0.09154653,"domain_scores_codex":[0.9987747,0.0006287581,0.0000800213,0.0002923342,0.0001060117,0.0001182434],"domain_scores_gemma":[0.9912096,0.007247871,0.0001332998,0.000386468,0.0007844425,0.0002382341],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004532875,0.001274807,0.01923747,0.001676894,0.0006161733,0.0007589031,0.003227975,0.3525099,0.007103012,0.009879888,0.06708401,0.5320982],"study_design_scores_gemma":[0.0001385278,0.00024763,0.005931892,0.0002341683,0.0002746578,0.0001541969,0.0009208643,0.9639208,0.004901486,0.007328619,0.01585723,0.00008991114],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6336412,0.00533876,0.2803812,0.004294667,0.0020062,0.0007483124,0.01985396,0.02609695,0.02763876],"genre_scores_gemma":[0.8773968,0.001434007,0.08012144,0.0002771256,0.0002068078,0.0005529448,0.02635338,0.002227713,0.01142977],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08541442,"threshold_uncertainty_score":0.1698345,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04187459156628819,"score_gpt":0.3029099270173801,"score_spread":0.2610353354510919,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}