{"id":"W4313247739","doi":"10.23962/ajic.i30.13906","title":"A word embedding trained on South African news data","year":2022,"lang":"en","type":"article","venue":"The African Journal of Information and Communication (AJIC)","topic":"Topic Modeling","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Xanadu Quantum Technologies (Canada)","funders":"National Research Foundation; Department of Science and Innovation, South Africa","keywords":"Word embedding; Word2vec; Embedding; Word (group theory); Vocabulary; Computer science; Natural language processing; Artificial intelligence; Representation (politics); Information retrieval; Mathematics; Linguistics; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001778367,0.00008510858,0.0001362324,0.0001962351,0.000565095,0.0002342318,0.003489799,0.00001480963,0.00002757819],"category_scores_gemma":[0.000128,0.00006273373,0.00003890036,0.0003851591,0.00004571973,0.001414292,0.001191419,0.0004156721,0.00000531231],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006230126,"about_ca_system_score_gemma":0.0001373814,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001841364,"about_ca_topic_score_gemma":0.000003309425,"domain_scores_codex":[0.9985043,0.0003158415,0.0005649665,0.00007330934,0.0004155364,0.0001259899],"domain_scores_gemma":[0.9973447,0.0001926598,0.0007783705,0.001510576,0.0001027753,0.00007087836],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001571493,0.00008839092,0.0001052421,0.0000132184,0.00008034658,0.000001850304,0.1540538,0.02432308,0.00002895669,0.1546854,0.004151834,0.6623107],"study_design_scores_gemma":[0.0013188,0.0002967916,0.0007943236,0.00004764751,0.00002921374,0.0002278253,0.07210752,0.7483456,0.00001671582,0.008187322,0.168357,0.0002712497],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06143223,0.0006216198,0.8402833,0.07245996,0.0003785986,0.0004503063,0.00003590652,0.0001224024,0.02421567],"genre_scores_gemma":[0.9770539,0.00006698195,0.02139685,0.001399506,0.00002448044,0.000006273191,0.000008986156,0.000003433648,0.0000395521],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9156217,"threshold_uncertainty_score":0.6484973,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05390517709744123,"score_gpt":0.2729152722790952,"score_spread":0.219010095181654,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}