{"id":"W4313247739","doi":"10.23962/ajic.i30.13906","title":"A word embedding trained on South African news data","year":2022,"lang":"en","type":"article","venue":"The African Journal of Information and Communication (AJIC)","topic":"Topic Modeling","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Xanadu Quantum Technologies (Canada)","funders":"National Research Foundation; Department of Science and Innovation, South Africa","keywords":"Word embedding; Word2vec; Embedding; Word (group theory); Vocabulary; Computer science; Natural language processing; Artificial intelligence; Representation (politics); Information retrieval; Mathematics; Linguistics; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001627702,0.00109252,0.0004054942,0.001058154,0.0003414465,0.0006045569,0.0003962933,0.0005536516,0.001814991],"category_scores_gemma":[0.006147055,0.0002147777,0.0006630364,0.0009158873,0.0003713695,0.001654685,0.000840157,0.001235829,0.001851546],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004565073,"about_ca_system_score_gemma":0.0006094266,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00525716,"about_ca_topic_score_gemma":0.006842682,"domain_scores_codex":[0.9992693,0.0003029258,0.00005772812,0.0001848431,0.00010981,0.0000752829],"domain_scores_gemma":[0.9978185,0.001151344,0.00008778314,0.0003648112,0.0005121832,0.00006538219],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002259681,0.00138292,0.02399746,0.0012616,0.0006077305,0.0006750356,0.001235221,0.1470093,0.03975912,0.003394475,0.04958943,0.7288279],"study_design_scores_gemma":[0.0001832831,0.001277132,0.0230106,0.0002094868,0.0002289748,0.000511263,0.001139784,0.8801639,0.05444667,0.003755838,0.03496689,0.000106133],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8950765,0.003017538,0.07499894,0.0009989032,0.001057888,0.0004768839,0.01229143,0.00395378,0.008128207],"genre_scores_gemma":[0.8599575,0.001110417,0.08807345,0.0002273318,0.0001958315,0.0003126408,0.0411082,0.0002419913,0.008772618],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00525716,"threshold_uncertainty_score":0.0104531,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05390517709744123,"score_gpt":0.2729152722790952,"score_spread":0.219010095181654,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}