{"id":"W2914180170","doi":"10.1093/llc/fqy074","title":"Toward Kurdish language processing: Experiments in collecting and processing the AsoSoft text corpus","year":2018,"lang":"en","type":"article","venue":"Digital Scholarship in the Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Perplexity; Computer science; Natural language processing; Text corpus; Artificial intelligence; n-gram; Language model; Text processing; Annotation; Zipf's law; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0008730627,0.0002225398,0.0001785389,0.0001824002,0.0006365174,0.005703764,0.001578075,0.00008062911,0.000003429829],"category_scores_gemma":[0.0005046194,0.0001400053,0.00003065709,0.0006656427,0.0004661063,0.003330587,0.0004519852,0.0005242678,0.000006191537],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009184306,"about_ca_system_score_gemma":0.0001008841,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005470035,"about_ca_topic_score_gemma":0.000173764,"domain_scores_codex":[0.9983539,0.0001274635,0.0002954564,0.000387596,0.0004253958,0.0004102064],"domain_scores_gemma":[0.9992105,0.0001719594,0.0001493133,0.0003332476,0.0001082427,0.00002678663],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00005801242,0.0002471551,0.01774142,0.0002782719,0.00001359359,0.0002055389,0.3857762,9.665222e-7,0.001374891,0.02027013,0.000170365,0.5738635],"study_design_scores_gemma":[0.00446305,0.002112551,0.06687786,0.009547178,0.00006515525,0.002206858,0.2055522,0.007596527,0.1029547,0.5817071,0.01109483,0.005822022],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9544082,0.02488184,0.001893705,0.000858577,0.000163386,0.0006727606,0.000003834416,0.0005728214,0.0165449],"genre_scores_gemma":[0.9956276,0.000004436905,0.002899292,0.0008811026,0.0001112683,0.0000678461,0.000002092001,0.0000194856,0.0003869228],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5680414,"threshold_uncertainty_score":0.9953284,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06108343550245978,"score_gpt":0.3139606671856046,"score_spread":0.2528772316831449,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}