{"id":"W1994871456","doi":"10.1109/icmla.2013.117","title":"Personalized Spam Filtering with Natural Language Attributes","year":2013,"lang":"en","type":"article","venue":"","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Naive Bayes classifier; Random forest; Machine learning; Support vector machine; Artificial intelligence; Classifier (UML); Benchmark (surveying); The Internet; Filter (signal processing); Bag-of-words model; F1 score; Data mining; Natural language processing; World Wide Web","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00167394,0.000569273,0.0007902349,0.001844731,0.0004841236,0.0009489435,0.0004454453,0.0008313924,0.001144032],"category_scores_gemma":[0.005320627,0.000209393,0.0005932788,0.001452772,0.0002873164,0.001641881,0.0004375043,0.0007307237,0.0009720748],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000476284,"about_ca_system_score_gemma":0.0006467639,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0025668,"about_ca_topic_score_gemma":0.00400131,"domain_scores_codex":[0.9986448,0.0004310412,0.00008869533,0.0002469525,0.0004880202,0.0001005441],"domain_scores_gemma":[0.9965929,0.001799937,0.000327788,0.000549957,0.0006650985,0.00006431732],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009744881,0.001094409,0.03272158,0.0003892724,0.0002404194,0.0003236903,0.0003086878,0.06903542,0.03914734,0.002668981,0.01215606,0.8409396],"study_design_scores_gemma":[0.00006863655,0.0005606081,0.02940646,0.00003670593,0.0001290394,0.0005755397,0.0001687413,0.9013937,0.04535948,0.008512552,0.01370192,0.00008677673],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6139124,0.002018723,0.3581264,0.0006977996,0.0001571728,0.0003223096,0.001921622,0.01613669,0.006706924],"genre_scores_gemma":[0.8565198,0.0003331281,0.1373775,0.0002218143,0.0001342951,0.0001190875,0.002705138,0.0001059267,0.00248329],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0025668,"threshold_uncertainty_score":0.00885272,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007964234677811353,"score_gpt":0.2023670910412849,"score_spread":0.1944028563634736,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}