{"id":"W2023740829","doi":"10.5555/2486788.2486927","title":"Automatic detection of performance deviations in the load testing of large scale systems","year":2013,"lang":"en","type":"article","venue":"","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":73,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; Queen's University","funders":"","keywords":"Computer science; Benchmark (surveying); False positive paradox; Overhead (engineering); Set (abstract data type); Data mining; Test set; Scale (ratio); Machine learning; Artificial intelligence; Reliability engineering; Real-time computing; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001661918,0.0006707796,0.0005861148,0.002632309,0.0004134834,0.0008319676,0.001038809,0.0006213378,0.0006294856],"category_scores_gemma":[0.01144124,0.000173889,0.0002716206,0.001782808,0.0004917061,0.0008641552,0.000709358,0.0006801034,0.0006238905],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007415413,"about_ca_system_score_gemma":0.0007199253,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001947447,"about_ca_topic_score_gemma":0.001958228,"domain_scores_codex":[0.9969738,0.0007201756,0.0002393628,0.0005171609,0.001379375,0.0001702144],"domain_scores_gemma":[0.9876122,0.004741168,0.00363345,0.001583164,0.002172206,0.0002578131],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0008023537,0.0009646627,0.2198394,0.0002399978,0.0001642107,0.0004407968,0.0003408635,0.1218148,0.08857729,0.0009125912,0.004266941,0.5616361],"study_design_scores_gemma":[0.00001072795,0.000238606,0.0548103,0.00001524964,0.00001801487,0.0002957612,0.0000581001,0.9019247,0.04091078,0.0008535289,0.0008334934,0.00003065643],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7935836,0.0002830941,0.1933946,0.0002276932,0.00004647235,0.00009322313,0.0005432487,0.009913377,0.001914791],"genre_scores_gemma":[0.9753734,0.0000283243,0.02375894,0.00002856093,0.00001574207,0.00002659797,0.0004199031,0.0000715464,0.0002770663],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002632309,"threshold_uncertainty_score":0.008789122,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01270176190692431,"score_gpt":0.226176354115137,"score_spread":0.2134745922082126,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}