{"id":"W4386566545","doi":"10.18653/v1/2023.fieldmatters-1.7","title":"Approaches to Corpus Creation for Low-Resource Language Technology: the Case of Southern Kurdish and Laki","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"National Science Foundation","keywords":"Standardization; Computer science; Resource (disambiguation); Task (project management); Face (sociological concept); World Wide Web; Language technology; Linguistics; Natural language processing; Natural language; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006147848,0.0006734914,0.0006557388,0.00635876,0.00422722,0.005312899,0.002171075,0.001320117,0.005864757],"category_scores_gemma":[0.01542893,0.0007865165,0.0005552254,0.005811826,0.003684368,0.005571195,0.006199506,0.001917357,0.002168422],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002212956,"about_ca_system_score_gemma":0.004322235,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007654131,"about_ca_topic_score_gemma":0.0177871,"domain_scores_codex":[0.996057,0.002063604,0.000354451,0.0006720711,0.0006452408,0.0002076075],"domain_scores_gemma":[0.9880893,0.006594244,0.0007442597,0.002375334,0.001757346,0.0004395215],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003210646,0.0003634373,0.01465743,0.003221581,0.0001469553,0.005379211,0.07959749,0.01068171,0.05209921,0.1807366,0.03643278,0.6163625],"study_design_scores_gemma":[0.0001209196,0.000200254,0.02401809,0.001119578,0.0001277398,0.004864444,0.06754848,0.05471457,0.03261847,0.09746404,0.716891,0.0003123341],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2231802,0.006440341,0.6665636,0.01258533,0.0007840885,0.002163388,0.005659776,0.006830354,0.07579295],"genre_scores_gemma":[0.2840302,0.001317215,0.6956576,0.0003977107,0.0001542804,0.0009399059,0.006479033,0.001061695,0.009962349],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007654131,"threshold_uncertainty_score":0.03251332,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03008579129693851,"score_gpt":0.2709553102570255,"score_spread":0.240869518960087,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}