{"id":"W4283080616","doi":"10.1109/icse-seip55303.2022.9793941","title":"The Impact of Flaky Tests on Historical Test Prioritization on Chrome","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Prioritization; Blocking (statistics); Pipeline (software); Computer science; Test (biology); Reliability engineering; Replication (statistics); Work (physics); Regression testing; Software; Engineering; Software development; Operating system; Computer network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01266691,0.001418708,0.000815942,0.002581785,0.001027429,0.002860635,0.002861283,0.001217001,0.001802754],"category_scores_gemma":[0.1105766,0.0009334614,0.0007150696,0.002132681,0.001830224,0.004787216,0.001614843,0.00257569,0.000714478],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002753046,"about_ca_system_score_gemma":0.002605725,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02291926,"about_ca_topic_score_gemma":0.02315117,"domain_scores_codex":[0.9834564,0.004501275,0.001081922,0.003559869,0.006347861,0.001052776],"domain_scores_gemma":[0.870997,0.08158273,0.005693791,0.02625155,0.0132805,0.002194553],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002464178,0.001112138,0.158509,0.0007542929,0.0004018586,0.0004040074,0.0009556048,0.1657184,0.0321301,0.007607119,0.02819412,0.6017492],"study_design_scores_gemma":[0.0003583132,0.001797203,0.06837135,0.0001894009,0.000239745,0.0007322407,0.0004920764,0.8443345,0.06308258,0.005818816,0.01434231,0.0002414922],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8164254,0.003323634,0.1068947,0.001941065,0.0004439073,0.0003778047,0.001620436,0.05660729,0.01236577],"genre_scores_gemma":[0.9000005,0.0002805858,0.0936924,0.0004391402,0.00004585141,0.00007810759,0.001687243,0.002030629,0.001745532],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02291926,"threshold_uncertainty_score":0.06698984,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01932153530382496,"score_gpt":0.2868846522061037,"score_spread":0.2675631169022787,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}