{"id":"W4378508602","doi":"10.48550/arxiv.2305.13989","title":"MasakhaPOS: Part-of-Speech Tagging for Typologically Diverse African Languages","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"DeepMind; International Development Research Centre; Rockefeller Foundation","keywords":"Computer science; Conditional random field; Natural language processing; Artificial intelligence; Field (mathematics); Transfer (computing); Baseline (sea); Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001256755,0.0009539683,0.0004710475,0.00316393,0.001537758,0.0009588689,0.0009843056,0.0008444014,0.007711974],"category_scores_gemma":[0.003022681,0.0003957768,0.0005854791,0.002679763,0.0005458063,0.002146319,0.002890665,0.00105277,0.00550438],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005502832,"about_ca_system_score_gemma":0.001247899,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00575401,"about_ca_topic_score_gemma":0.01121363,"domain_scores_codex":[0.9990792,0.0002432301,0.0001047245,0.0003080798,0.0001422988,0.0001224495],"domain_scores_gemma":[0.9985697,0.0004766044,0.0001732179,0.0004244366,0.0002134855,0.0001425633],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00355805,0.0006234563,0.08546638,0.004516135,0.0004109147,0.00298944,0.00624775,0.008364357,0.08542023,0.01639655,0.4141675,0.3718393],"study_design_scores_gemma":[0.0004838259,0.0003764294,0.1325895,0.0005821019,0.0002161821,0.003033051,0.007379992,0.03582497,0.05289459,0.02361453,0.7427409,0.0002638727],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.3894548,0.002099744,0.05372833,0.001224646,0.0008540956,0.0006000164,0.5088629,0.01967593,0.02349957],"genre_scores_gemma":[0.2469644,0.0005151677,0.07336514,0.0004152351,0.0001100176,0.001109705,0.670683,0.001323788,0.005513563],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007711974,"threshold_uncertainty_score":0.02579916,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1160671087584835,"score_gpt":0.2457228766961179,"score_spread":0.1296557679376344,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}