{"id":"W3093186001","doi":"10.1109/csp51677.2021.9357595","title":"SpaML: a Bimodal Ensemble Learning Spam Detector based on NLP Techniques","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université Laval","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Artificial intelligence; Computer science; Classifier (UML); Natural language processing; Ensemble learning; tf–idf; Set (abstract data type); Detector; Pattern recognition (psychology); Machine learning; Term (time)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0005654214,0.0003805887,0.0003677303,0.0003701942,0.0001984815,0.001001367,0.001059764,0.0004780931,0.0001223265],"category_scores_gemma":[0.0001952085,0.000370825,0.0002680915,0.0004028991,0.00002746538,0.0002056935,0.000997085,0.001438294,0.0000560547],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001692301,"about_ca_system_score_gemma":0.0003185507,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006155977,"about_ca_topic_score_gemma":0.00009432889,"domain_scores_codex":[0.9973915,0.0002565612,0.000306288,0.001113188,0.0005526621,0.0003798137],"domain_scores_gemma":[0.9980694,0.0001996634,0.0001961813,0.001237529,0.0001645396,0.000132703],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001630921,0.0008471224,0.003228429,0.0008831847,0.0002204802,0.0006135713,0.00202545,0.07400957,0.06694099,0.006878035,0.004311451,0.8398786],"study_design_scores_gemma":[0.0002015073,0.0004392116,0.0006727255,0.0004398837,0.00001918284,0.00001817083,0.00002428166,0.6425323,0.3453957,0.001538905,0.007910038,0.0008080772],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02504635,0.00005765802,0.9527205,0.0006825341,0.001503984,0.000317421,0.000001169366,0.002288626,0.01738179],"genre_scores_gemma":[0.8643171,0.00001594311,0.1338891,0.0005821188,0.0003226915,0.00008740615,0.0000130034,0.0000353557,0.000737241],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8392708,"threshold_uncertainty_score":0.9998744,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01738504715447093,"score_gpt":0.244266886106616,"score_spread":0.2268818389521451,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}