{"id":"W1990295648","doi":"10.1145/1571941.1572107","title":"On the relative age of spam and ham training samples for email filtering","year":2009,"lang":"en","type":"article","venue":"","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Sample (material); Filter (signal processing); Artificial intelligence; Computer vision","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001999041,0.00004182611,0.00005803004,0.00002490486,0.00008108659,0.00004764034,0.0001226517,0.00001831733,0.000004623711],"category_scores_gemma":[0.00009672778,0.00002682401,0.00002352745,0.00006735003,0.00001526319,0.0001432018,0.00001656882,0.0000404523,6.01676e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000004759171,"about_ca_system_score_gemma":0.000005170126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001463603,"about_ca_topic_score_gemma":0.00000974371,"domain_scores_codex":[0.9996769,0.00001539215,0.00006857865,0.0001078563,0.00005827443,0.00007304227],"domain_scores_gemma":[0.9994466,0.0003749598,0.0000303044,0.0001209583,0.00001246803,0.00001473304],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00001182854,0.00001202815,0.0000251785,0.000004638307,0.000008497033,8.747218e-7,0.008129079,0.00006370838,0.009677717,0.8943887,0.0003234569,0.08735429],"study_design_scores_gemma":[0.0006261203,0.00161021,0.02372775,0.0001284664,0.00001120986,0.00001410076,0.0004714188,0.0739251,0.04745052,0.8480042,0.003750338,0.0002805423],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1285665,0.0000134298,0.8654833,0.001301664,0.00007629535,0.00009195681,7.700496e-7,0.00004186295,0.004424162],"genre_scores_gemma":[0.9632117,0.000001653357,0.03631104,0.0003051496,0.00002232241,0.000002823666,3.243045e-7,0.000001528645,0.0001434427],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8346452,"threshold_uncertainty_score":0.1093851,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08240936578597065,"score_gpt":0.2611598832613626,"score_spread":0.1787505174753919,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}