{"id":"W4385409983","doi":"10.21203/rs.3.rs-3204260/v1","title":"Evaluating Supervised Machine Learning Models for Zero-Day Phishing Attack Detection: A Comprehensive Study","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Phishing; Computer science; Machine learning; Dimensionality reduction; Artificial intelligence; Data mining; Dimension (graph theory); Zero (linguistics); The Internet; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.008355834,0.0004875207,0.0006487307,0.001066374,0.001857552,0.001936086,0.002193206,0.0004126897,0.00001339604],"category_scores_gemma":[0.002116238,0.0005061068,0.0003618828,0.001507846,0.00007748012,0.0007452409,0.004360726,0.003726401,0.00008984569],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005730418,"about_ca_system_score_gemma":0.0004048207,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001675802,"about_ca_topic_score_gemma":0.0004606151,"domain_scores_codex":[0.9903813,0.003167463,0.0006799453,0.001959219,0.00267413,0.001137963],"domain_scores_gemma":[0.9922673,0.003361238,0.0002454198,0.001579409,0.002274258,0.0002724322],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001978007,0.0003766175,0.0009540754,0.00160741,0.0003437458,0.00007242563,0.02018842,0.8584816,0.002718898,0.0002473274,0.0005680087,0.1142437],"study_design_scores_gemma":[0.001072613,0.001684775,0.001232573,0.0004852388,0.00002724702,0.000005870022,0.0009914251,0.9817415,0.0003881321,0.01147122,0.0004229705,0.0004763852],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2458126,0.0007710439,0.7433139,0.0006690497,0.002106482,0.005635949,0.00004235865,0.001519833,0.0001287255],"genre_scores_gemma":[0.9891647,0.00007871322,0.007396507,0.00002471191,0.0006738532,0.001777649,0.00007519434,0.0001260566,0.0006825862],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7433521,"threshold_uncertainty_score":0.9997391,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4647948181065284,"score_gpt":0.478203365313575,"score_spread":0.01340854720704659,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}