{"id":"W2012114007","doi":"10.1007/s10207-012-0178-1","title":"Evaluation in the absence of absolute ground truth: toward reliable evaluation methodology for scan detectors","year":2012,"lang":"en","type":"article","venue":"International Journal of Information Security","topic":"Network Security and Intrusion Detection","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University","funders":"","keywords":"Ground truth; Computer science; Detector; Intrusion detection system; Data mining; Artificial intelligence; Intrusion; Common ground; Machine learning; Telecommunications","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02957709,0.001754522,0.00183906,0.003645961,0.0007804274,0.004868468,0.002815617,0.003144523,0.001445072],"category_scores_gemma":[0.09904858,0.0006782105,0.0008252863,0.001808094,0.001867914,0.005956382,0.002747591,0.001738647,0.0007154795],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001547076,"about_ca_system_score_gemma":0.002639463,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002671617,"about_ca_topic_score_gemma":0.003911218,"domain_scores_codex":[0.9709442,0.01306491,0.002348909,0.003226259,0.009529969,0.0008856549],"domain_scores_gemma":[0.9267055,0.03906287,0.006199378,0.008248992,0.01881934,0.0009639084],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001941319,0.0008193567,0.06211108,0.001462558,0.001137452,0.0003812345,0.000661727,0.2131967,0.0506037,0.02003551,0.008730453,0.6389189],"study_design_scores_gemma":[0.00005740645,0.0008461801,0.007752341,0.0001557031,0.0001971367,0.000389282,0.000237717,0.9406676,0.03298821,0.0138798,0.00276315,0.00006543149],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0844642,0.0008592829,0.9088401,0.0003420698,0.00008522614,0.0003556335,0.0004921581,0.001914947,0.002646293],"genre_scores_gemma":[0.6917683,0.0003613996,0.3043066,0.000222334,0.00008099839,0.0003238311,0.001216781,0.0003890497,0.001330727],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02957709,"threshold_uncertainty_score":0.1564205,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0960749425709773,"score_gpt":0.3628560966939349,"score_spread":0.2667811541229576,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}