{"id":"W2251433042","doi":"10.3115/v1/w14-5904","title":"Automatic Identification of Arabic Language Varieties and Dialects in Social Media","year":2014,"lang":"en","type":"article","venue":"","topic":"Authorship Attribution and Profiling","field":"Computer Science","cited_by":90,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Montréal","funders":"","keywords":"Arabic; Computer science; Identification (biology); Natural language processing; Social media; Language identification; Linguistics; Artificial intelligence; Natural language; World Wide Web; Botany","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008752725,0.0005792789,0.0003818967,0.004159898,0.000706622,0.0012393,0.0003410477,0.0005291293,0.001470246],"category_scores_gemma":[0.004146681,0.0001582098,0.0003540178,0.001660319,0.0002497002,0.001869543,0.0009432223,0.0005094813,0.002242686],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000355879,"about_ca_system_score_gemma":0.0003263717,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002623127,"about_ca_topic_score_gemma":0.004596252,"domain_scores_codex":[0.9991266,0.0003011423,0.00007853698,0.0002122599,0.0001793822,0.0001021814],"domain_scores_gemma":[0.9975266,0.0009359519,0.0003846111,0.000273519,0.0007195079,0.0001598517],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001025122,0.0004986657,0.2708468,0.000606929,0.00014157,0.0008975473,0.003802615,0.003560876,0.05054202,0.003435937,0.02145712,0.6431848],"study_design_scores_gemma":[0.00006310661,0.0003746186,0.4543249,0.0003347096,0.0001885753,0.002926136,0.01418138,0.4061882,0.05882713,0.0143239,0.04807148,0.0001958421],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9358654,0.001292436,0.0455699,0.0006704699,0.0002643844,0.000269839,0.006366311,0.001561825,0.008139399],"genre_scores_gemma":[0.9479809,0.0003899872,0.04172231,0.00008631561,0.0001005767,0.00009137209,0.006239462,0.00004986307,0.003339113],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004159898,"threshold_uncertainty_score":0.005215764,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01806990496924178,"score_gpt":0.2669399753216908,"score_spread":0.248870070352449,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}