{"id":"W4409231750","doi":"10.14778/3712221.3712227","title":"How Reliable are Streams? End-to-End Processing-Guarantee Validation and Performance Benchmarking of Stream Processing Systems","year":2024,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Distributed systems and fault tolerance","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Benchmarking; Stream processing; STREAMS; Computer science; End-to-end principle; Real-time computing; Distributed computing; Computer network; Business","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005555667,0.0002495022,0.0003642247,0.0001636378,0.0001998865,0.0009788572,0.0007452754,0.00007324849,0.00000113775],"category_scores_gemma":[0.00003397429,0.0001751386,0.00007086747,0.0008868147,0.00007417287,0.001341535,0.000301007,0.00016445,0.000001105878],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009727423,"about_ca_system_score_gemma":0.00009417335,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005230526,"about_ca_topic_score_gemma":0.00000139963,"domain_scores_codex":[0.9979001,0.000008853878,0.0004954385,0.0005522479,0.0006990411,0.0003442678],"domain_scores_gemma":[0.9988027,0.00002388753,0.0004905125,0.0002015015,0.00041044,0.00007096711],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001032953,0.0004525361,0.03799323,0.03154788,0.0002286608,0.000006713861,0.01131813,0.002476626,0.08028921,0.02823682,0.003757865,0.803589],"study_design_scores_gemma":[0.0009094339,0.0006496123,0.006404999,0.02336132,0.0001401316,0.0001400024,0.003120309,0.7028134,0.2507963,0.0006634847,0.01011854,0.0008824221],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9836027,0.005281876,0.006277245,0.001212514,0.0007000373,0.001057423,0.00004012221,0.0001837396,0.001644323],"genre_scores_gemma":[0.9977772,0.00009234342,0.001506434,0.0000157561,0.000113505,0.00009027287,0.000002367603,0.00001783581,0.0003843162],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8027066,"threshold_uncertainty_score":0.9439142,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01200811440295121,"score_gpt":0.220075315563754,"score_spread":0.2080672011608028,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}