{"id":"W3173053558","doi":"10.48550/arxiv.2105.13573","title":"Investigating Code-Mixed Modern Standard Arabic-Egyptian to English Machine Translation","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Machine translation; Computer science; Natural language processing; Arabic; Artificial intelligence; Scratch; Task (project management); Transformer; Language model; Code-switching; Code (set theory); Context (archaeology); Modern Standard Arabic; Speech recognition; Programming language; Linguistics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002179348,0.001108103,0.0005948654,0.001048346,0.000837518,0.001517577,0.0007463216,0.001130048,0.003021332],"category_scores_gemma":[0.007986167,0.0003665377,0.0006102581,0.001246203,0.0007449144,0.002142988,0.001424263,0.001447715,0.001789124],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001169907,"about_ca_system_score_gemma":0.001239399,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01874206,"about_ca_topic_score_gemma":0.02460765,"domain_scores_codex":[0.9989145,0.0004680261,0.0000818634,0.0002998714,0.00013796,0.00009781969],"domain_scores_gemma":[0.9957101,0.002491203,0.0001790511,0.0005550961,0.0008771729,0.0001873686],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002884962,0.001441716,0.04629607,0.001733122,0.0007037662,0.002855047,0.003791121,0.3719522,0.05263371,0.02226078,0.03436445,0.4590831],"study_design_scores_gemma":[0.0001663005,0.0005256853,0.00891556,0.00009833099,0.0001295112,0.0005957144,0.001536291,0.9185112,0.04168774,0.01185128,0.01589921,0.00008319227],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9153649,0.001427105,0.06343307,0.001051809,0.000280558,0.0001782797,0.00211813,0.003952894,0.01219329],"genre_scores_gemma":[0.9483985,0.0002744643,0.03926381,0.0003230783,0.00005900257,0.0001321336,0.006174089,0.0004168452,0.004958139],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01874206,"threshold_uncertainty_score":0.03726596,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05796563702836924,"score_gpt":0.2112129708741395,"score_spread":0.1532473338457703,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}