{"id":"W2914180170","doi":"10.1093/llc/fqy074","title":"Toward Kurdish language processing: Experiments in collecting and processing the AsoSoft text corpus","year":2018,"lang":"en","type":"article","venue":"Digital Scholarship in the Humanities","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Perplexity; Computer science; Natural language processing; Text corpus; Artificial intelligence; n-gram; Language model; Text processing; Annotation; Zipf's law; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004656058,0.002016712,0.001347173,0.001708873,0.002115683,0.001710085,0.001951316,0.00177801,0.004908395],"category_scores_gemma":[0.01668897,0.0004666741,0.0009493587,0.003256578,0.001486233,0.003629856,0.002381562,0.002097729,0.004068428],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001028248,"about_ca_system_score_gemma":0.001565065,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01276032,"about_ca_topic_score_gemma":0.01016511,"domain_scores_codex":[0.9952123,0.00221894,0.0004454875,0.001152771,0.0006797513,0.0002907148],"domain_scores_gemma":[0.987661,0.008897391,0.0002137806,0.001610059,0.001197214,0.0004205847],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.005892112,0.007285814,0.01200676,0.004774882,0.0005566829,0.004308332,0.00907959,0.04896782,0.05801672,0.006137922,0.1106426,0.7323307],"study_design_scores_gemma":[0.002733077,0.003840225,0.0444665,0.0005304298,0.0006236986,0.003649428,0.0210625,0.6006711,0.1646051,0.01526966,0.1419117,0.0006366257],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9150215,0.001570247,0.04081346,0.001570601,0.0006203067,0.001617123,0.01095542,0.01187757,0.01595371],"genre_scores_gemma":[0.7203261,0.000848819,0.1988744,0.001106369,0.0002202468,0.002025897,0.066419,0.001412339,0.008766886],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01276032,"threshold_uncertainty_score":0.02537209,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06108343550245978,"score_gpt":0.3139606671856046,"score_spread":0.2528772316831449,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}