{"id":"W1994871456","doi":"10.1109/icmla.2013.117","title":"Personalized Spam Filtering with Natural Language Attributes","year":2013,"lang":"en","type":"article","venue":"","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Naive Bayes classifier; Random forest; Machine learning; Support vector machine; Artificial intelligence; Classifier (UML); Benchmark (surveying); The Internet; Filter (signal processing); Bag-of-words model; F1 score; Data mining; Natural language processing; World Wide Web","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00005195454,0.00006907016,0.00006376908,0.00003419486,0.0000646182,0.0002330487,0.0002558084,0.00001981609,0.0002318501],"category_scores_gemma":[0.00001130614,0.00004638722,0.00002320333,0.0001522748,0.00001663063,0.0005596412,0.00006230272,0.00007521663,0.0001916042],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001496047,"about_ca_system_score_gemma":0.00000896384,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004133171,"about_ca_topic_score_gemma":0.00002385628,"domain_scores_codex":[0.9994925,0.00001487483,0.00005375273,0.0001632933,0.0001282244,0.0001473327],"domain_scores_gemma":[0.9996741,0.00003325313,0.00002073971,0.0001941159,0.00003599294,0.00004179274],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007941898,0.0002276303,0.02127601,0.0001388148,0.0002657872,0.0001384334,0.02596825,0.000215889,0.4030369,0.0711663,0.06396571,0.4135208],"study_design_scores_gemma":[0.003531797,0.0008319253,0.1120553,0.0001540724,0.00002576081,0.000450659,0.001562006,0.6402726,0.2144802,0.002075966,0.0226841,0.001875562],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.613253,0.0002824265,0.3788024,0.001762112,0.000413591,0.0001644943,5.281401e-7,0.0005942595,0.004727168],"genre_scores_gemma":[0.9656694,0.000001179115,0.03086361,0.0003273275,0.00006259091,0.00001107834,0.000001384471,0.000004117926,0.003059316],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6400567,"threshold_uncertainty_score":0.2538595,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007964234677811353,"score_gpt":0.2023670910412849,"score_spread":0.1944028563634736,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}