{"id":"W4292621974","doi":"10.1145/3546157.3546173","title":"N-gram and Word2Vec Feature Engineering Approaches for Spam Recognition on Some Influential Twitter Topics in Saudi Arabia","year":2022,"lang":"en","type":"article","venue":"","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Word2vec; Social media; Computer science; Sentiment analysis; Artificial intelligence; Arabic; Machine learning; Feature engineering; Feature (linguistics); Random forest; n-gram; Data science; World Wide Web; Deep learning; Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001790119,0.00007895331,0.00007690679,0.000122679,0.00009076672,0.0001021741,0.0001388156,0.0000420061,0.000007099504],"category_scores_gemma":[0.00002077276,0.00007924821,0.0000290617,0.0001756364,0.000005264764,0.0002341326,0.00009428197,0.0001934498,0.000001619309],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004421988,"about_ca_system_score_gemma":0.000008755096,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001897485,"about_ca_topic_score_gemma":0.000009675671,"domain_scores_codex":[0.9994042,0.00002198642,0.00008430934,0.0002366537,0.0001130547,0.0001398171],"domain_scores_gemma":[0.9997662,0.00004118782,0.00002536701,0.0001312126,0.000008685877,0.0000273795],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002271811,0.0003675065,0.004191471,0.0002249665,0.0000611681,0.00001987112,0.005243368,0.02627116,0.001875797,0.03487059,0.007406471,0.9192404],"study_design_scores_gemma":[0.002932166,0.001255127,0.05192494,0.00007363136,0.00002154843,0.0000674479,0.0001897292,0.827392,0.008727127,0.03964826,0.06675429,0.001013779],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9090155,0.00007473196,0.08567255,0.003201135,0.001052042,0.0004783964,0.00000373559,0.0001848812,0.0003170024],"genre_scores_gemma":[0.9797564,0.000004288394,0.0185244,0.0008249229,0.0002371784,0.0001989441,0.00001138927,0.00001011093,0.0004324352],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9182267,"threshold_uncertainty_score":0.3231648,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03637850094376847,"score_gpt":0.2177338790444393,"score_spread":0.1813553781006708,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}