{"id":"W3096465442","doi":"10.1002/spe.2929","title":"Root causing, detecting, and fixing flaky tests: State of the art and future roadmap","year":2020,"lang":"en","type":"article","venue":"Software Practice and Experience","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"ca_institutions":"Brandon University; University of Guelph; University of Guelph-Humber","funders":"","keywords":"Pace; Root cause; Software deployment; Test (biology); Field (mathematics); Computer science; Root (linguistics); Engineering; Data science; Software; Engineering management; Software engineering; Operations management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01412195,0.001682686,0.001771524,0.009353181,0.0007912846,0.004579197,0.003575745,0.003528364,0.005459324],"category_scores_gemma":[0.02560542,0.000776199,0.001950016,0.005513124,0.002280566,0.009019743,0.001854788,0.002737615,0.001748925],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002128914,"about_ca_system_score_gemma":0.007687964,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005002959,"about_ca_topic_score_gemma":0.003848866,"domain_scores_codex":[0.9925537,0.002310378,0.000916176,0.001238053,0.002572442,0.000409303],"domain_scores_gemma":[0.9281811,0.04969717,0.004338666,0.001884207,0.0147909,0.001107917],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001191212,0.0002233976,0.00283082,0.01824634,0.0001050101,0.00009152765,0.0003739688,0.001510557,0.001193223,0.01000656,0.008221201,0.9570782],"study_design_scores_gemma":[0.0001109489,0.002025121,0.0121989,0.1072904,0.002073843,0.002092829,0.006329282,0.01926878,0.008748109,0.06053438,0.7789053,0.0004219422],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.00214165,0.9766771,0.01191389,0.005028501,0.0004713175,0.0001023746,0.0001044644,0.0002280186,0.003332792],"genre_scores_gemma":[0.03099062,0.9352823,0.02896993,0.002194962,0.0007550213,0.0001843784,0.0004171135,0.00007140089,0.001134282],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.01412195,"threshold_uncertainty_score":0.07468492,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01675070680961436,"score_gpt":0.2725751747409247,"score_spread":0.2558244679313103,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}