{"id":"W2962724530","doi":"10.18653/v1/p17-2095","title":"Challenging Language-Dependent Segmentation for Arabic: An\\n Application to Machine Translation and Part-of-Speech Tagging","year":2017,"lang":"en","type":"article","venue":"Figshare","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Arabic; Computer science; Machine translation; Natural language processing; Artificial intelligence; Computational linguistics; Speech recognition; Segmentation; Volume (thermodynamics); Speech translation; Linguistics; Translation (biology); Philosophy","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00119743,0.00149327,0.0006885231,0.001021348,0.001415621,0.001792445,0.0009254722,0.002222164,0.007649945],"category_scores_gemma":[0.0049885,0.000324157,0.0005795134,0.001834385,0.0006848762,0.001524242,0.001121877,0.001236663,0.004720965],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008333971,"about_ca_system_score_gemma":0.0009723693,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007461717,"about_ca_topic_score_gemma":0.01039042,"domain_scores_codex":[0.9992566,0.0002779569,0.00004955829,0.0002071295,0.0001474399,0.00006137095],"domain_scores_gemma":[0.9972204,0.001595873,0.0001295397,0.0003956365,0.000524339,0.0001342027],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001250029,0.0004764952,0.007883281,0.00110268,0.000149866,0.002932857,0.0009866705,0.1051764,0.09546546,0.0123515,0.05381642,0.7184082],"study_design_scores_gemma":[0.00005809442,0.0001839172,0.00411982,0.00004728911,0.00004760668,0.0009445014,0.0005393993,0.8895593,0.07160914,0.01390327,0.01892074,0.00006694545],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.464555,0.003666945,0.4737291,0.005623164,0.001285461,0.0004004386,0.003772247,0.02081811,0.02614949],"genre_scores_gemma":[0.6863377,0.001224643,0.292385,0.0007427487,0.0002910731,0.0001910219,0.005749227,0.001143479,0.01193499],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007649945,"threshold_uncertainty_score":0.02559167,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04363071102748368,"score_gpt":0.3350684569612692,"score_spread":0.2914377459337856,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}