{"id":"W4377694488","doi":"10.1117/12.2674952","title":"Analysis and comparison of machine learning methods and improved SVM algorithm in spam classification","year":2023,"lang":"en","type":"article","venue":"","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Support vector machine; Machine learning; Naive Bayes classifier; Random forest; Categorization; Artificial intelligence; Statistical classification; Text categorization; Bag-of-words model; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008157843,0.00005424471,0.0001645257,0.0003895425,0.00004809536,0.00005631528,0.00009565016,0.00004034512,0.000003390499],"category_scores_gemma":[0.0000626214,0.00004885881,0.00002255691,0.001577419,0.00001971478,0.0001532658,0.00008562853,0.0001043641,9.967273e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000009361422,"about_ca_system_score_gemma":0.000005759769,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000744545,"about_ca_topic_score_gemma":0.0002167635,"domain_scores_codex":[0.9993084,0.0001424635,0.000171892,0.0002167787,0.00007435503,0.00008611793],"domain_scores_gemma":[0.9995592,0.0001757855,0.00007539295,0.0001352448,0.00002433122,0.00003003265],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002288755,0.00001300296,0.1510651,0.00000783035,0.00003174218,2.591423e-7,0.0008544683,0.0001916593,0.0153564,0.0003377627,0.000005494826,0.8321339],"study_design_scores_gemma":[0.00009168913,0.0000360715,0.2339358,0.000001680457,0.00001580078,4.302222e-7,0.00007438191,0.7617062,0.003790919,0.0002179749,0.00008772756,0.00004128228],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1179232,0.00009593226,0.881487,0.0001875614,0.00004854717,0.00004462939,3.848131e-7,0.00008621434,0.0001265254],"genre_scores_gemma":[0.7921156,0.00003462956,0.2077164,0.000008093444,0.000006661302,0.000003061979,0.000003652365,0.000002237746,0.0001095835],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8320927,"threshold_uncertainty_score":0.1992404,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04183190352523331,"score_gpt":0.3608106111180451,"score_spread":0.3189787075928118,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}