{"id":"W4411449762","doi":"10.1145/3729346","title":"CAShift: Benchmarking Log-Based Cloud Attack Detection under Normality Shift","year":2025,"lang":"en","type":"article","venue":"Proceedings of the ACM on software engineering.","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Cloud computing; Computer science; Normality; Benchmarking; Data mining; Construct (python library); Paradigm shift; Anomaly detection; Statistics; Mathematics; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002634072,0.002665178,0.001159045,0.003319465,0.0008881235,0.001615534,0.003031,0.001542414,0.001205498],"category_scores_gemma":[0.007795872,0.0004004013,0.00111195,0.002329947,0.001150779,0.00254777,0.001913921,0.001945943,0.001324951],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00177346,"about_ca_system_score_gemma":0.001888521,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01692883,"about_ca_topic_score_gemma":0.01848605,"domain_scores_codex":[0.9968932,0.0004766553,0.0003449534,0.0009801558,0.0008828715,0.0004221149],"domain_scores_gemma":[0.9964618,0.0009592849,0.0003816461,0.0008567582,0.0009525624,0.0003879798],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004222825,0.003952868,0.1259791,0.002661347,0.001189946,0.001053286,0.0005007572,0.3375867,0.01862217,0.004351208,0.1532878,0.3465918],"study_design_scores_gemma":[0.0002478261,0.000881858,0.02397524,0.00006933073,0.00006466557,0.0004690237,0.0002406323,0.9503839,0.01142337,0.001696821,0.01046339,0.00008401052],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8364463,0.006699526,0.05632481,0.001867175,0.002205693,0.001179109,0.03106352,0.05599587,0.008217947],"genre_scores_gemma":[0.8864237,0.0008609109,0.04521307,0.0005502438,0.0002052773,0.0002682903,0.06394218,0.0005053142,0.002030939],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01692883,"threshold_uncertainty_score":0.03366059,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01056494758060279,"score_gpt":0.232949269598238,"score_spread":0.2223843220176352,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}