{"id":"W7125517337","doi":"10.55492/v6i02.6736","title":"Mafoko: Structuring and Building Open Multilingual Terminologies for South African NLP","year":2025,"lang":"","type":"article","venue":"Journal of the Digital Humanities Association of Southern Africa","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Pretoria; International Development Research Centre; Nvidia","keywords":"Terminology; Interoperability; Bridging (networking); Consistency (knowledge bases); Structuring; Scalability; Machine translation","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005061758,0.001156553,0.0007296744,0.00669365,0.002470502,0.003444324,0.002122248,0.001518371,0.01048695],"category_scores_gemma":[0.02450108,0.0007536996,0.001679066,0.004379577,0.001545197,0.00756113,0.009805645,0.002870681,0.008413125],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001976848,"about_ca_system_score_gemma":0.005309802,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008651732,"about_ca_topic_score_gemma":0.01400715,"domain_scores_codex":[0.9965844,0.001312504,0.0005175198,0.0007164169,0.000646368,0.0002227421],"domain_scores_gemma":[0.9935399,0.002570625,0.000447346,0.002177174,0.0009631714,0.0003019059],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0007541061,0.0004178972,0.01175041,0.006438436,0.0002960799,0.002029785,0.0113842,0.01675734,0.03880847,0.1022708,0.4389578,0.3701347],"study_design_scores_gemma":[0.0002945795,0.0001268706,0.007259238,0.001089791,0.00008788932,0.0009806011,0.004633814,0.06030294,0.02036165,0.08003578,0.8246413,0.0001855657],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08334444,0.003067711,0.4747057,0.004804058,0.0009952382,0.003533954,0.3163069,0.07600853,0.03723339],"genre_scores_gemma":[0.0865445,0.0008288653,0.4794304,0.0005744016,0.00009030994,0.002803196,0.4203306,0.005701863,0.003695985],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01048695,"threshold_uncertainty_score":0.0350824,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02697340096356203,"score_gpt":0.2787776361997294,"score_spread":0.2518042352361674,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}