{"id":"W4286331442","doi":"10.1109/saner53432.2022.00116","title":"Phishing Kits Source Code Similarity Distribution: A Case Study","year":2022,"lang":"en","type":"article","venue":"2022 IEEE International Conference on Software Analysis, Evolution and Reengineering (SANER)","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa; IBM (Canada); Polytechnique Montréal","funders":"","keywords":"Phishing; Computer science; Identifier; Similarity (geometry); Source code; Code (set theory); World Wide Web; Credit card; Identification (biology); Computer security; Information retrieval; Artificial intelligence; The Internet; Programming language; Payment","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001437974,0.0004116094,0.000345227,0.003957459,0.001143267,0.0008601471,0.0007859965,0.001273768,0.0006531539],"category_scores_gemma":[0.007912459,0.0002667069,0.000427201,0.003338228,0.001092689,0.001152552,0.001246044,0.0007547019,0.0004040094],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005771792,"about_ca_system_score_gemma":0.0004447469,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003410618,"about_ca_topic_score_gemma":0.004682873,"domain_scores_codex":[0.9979324,0.0004376573,0.0001818617,0.0004489948,0.0007948243,0.0002042701],"domain_scores_gemma":[0.9893317,0.004847233,0.001862238,0.001537334,0.001958926,0.0004624805],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0009632059,0.001689906,0.7154924,0.0009218324,0.0003586363,0.03423992,0.01582984,0.01995615,0.04089225,0.004394726,0.00902249,0.1562387],"study_design_scores_gemma":[0.0001038713,0.001459184,0.7110501,0.0002583832,0.0003016901,0.05356459,0.01451032,0.117425,0.0745664,0.004267763,0.0222841,0.0002085915],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9949543,0.0001935397,0.003273336,0.00009812584,0.000008302881,0.00005545988,0.0005312166,0.0002223281,0.0006633233],"genre_scores_gemma":[0.9906884,0.0002370123,0.006755675,0.0000372677,0.00001471325,0.00003615232,0.001383105,0.00007864353,0.0007689715],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003957459,"threshold_uncertainty_score":0.007604837,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03310186504630544,"score_gpt":0.2763223032600144,"score_spread":0.243220438213709,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}