{"id":"W4377694488","doi":"10.1117/12.2674952","title":"Analysis and comparison of machine learning methods and improved SVM algorithm in spam classification","year":2023,"lang":"en","type":"article","venue":"","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Support vector machine; Machine learning; Naive Bayes classifier; Random forest; Categorization; Artificial intelligence; Statistical classification; Text categorization; Bag-of-words model; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003324969,0.0006356915,0.0009349055,0.002903975,0.000389066,0.001113335,0.0006548552,0.001003957,0.001283675],"category_scores_gemma":[0.007102664,0.0001788655,0.0008116644,0.002177577,0.0003341922,0.001714192,0.0003426936,0.0008899248,0.0005230054],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000676339,"about_ca_system_score_gemma":0.0007315915,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002817529,"about_ca_topic_score_gemma":0.00157567,"domain_scores_codex":[0.9967205,0.001059619,0.0002883036,0.0003163085,0.001413501,0.0002018848],"domain_scores_gemma":[0.9939449,0.002900469,0.0003536404,0.0003239556,0.002397395,0.0000795554],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008233065,0.000411624,0.01512857,0.0005992025,0.0002742386,0.0001750617,0.0002147752,0.1169574,0.008645186,0.004956761,0.004651241,0.8471627],"study_design_scores_gemma":[0.00002524864,0.0003733346,0.009765898,0.00005135713,0.0001012873,0.0001868132,0.0000999491,0.9759061,0.008169361,0.001720152,0.00357039,0.00003010143],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3110441,0.0189869,0.6557838,0.000951485,0.0007072454,0.000223008,0.0003010116,0.002105811,0.009896639],"genre_scores_gemma":[0.7770264,0.004201945,0.2137306,0.0001518216,0.0003135543,0.0001366012,0.0005875152,0.0000924118,0.003759039],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003324969,"threshold_uncertainty_score":0.01758432,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04183190352523331,"score_gpt":0.3608106111180451,"score_spread":0.3189787075928118,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}