{"id":"W7125939950","doi":"10.1109/ase63991.2025.00251","title":"Who’s to Blame? Rethinking the Brittleness of Automated Web GUI Testing from a Pragmatic Perspective","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Memorial University of Newfoundland; University of Waterloo","funders":"","keywords":"Automation; Perspective (graphical); Web application; Test (biology); Test case; Software testing; Brittleness","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05005581,0.000863236,0.0007574979,0.00371227,0.002118757,0.006817743,0.002747821,0.002171159,0.00183789],"category_scores_gemma":[0.2223689,0.0009710047,0.0008345832,0.001426218,0.007603667,0.009592277,0.004122521,0.003357083,0.0005028994],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002497811,"about_ca_system_score_gemma":0.004461699,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006031818,"about_ca_topic_score_gemma":0.01205926,"domain_scores_codex":[0.9519487,0.03064734,0.002125306,0.003723645,0.01008412,0.001470932],"domain_scores_gemma":[0.7816958,0.1604908,0.01405396,0.02270131,0.01890127,0.00215692],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000932763,0.0005983667,0.2198229,0.002203658,0.0003901333,0.005310093,0.05490788,0.03513744,0.04639974,0.1105516,0.02015424,0.5035912],"study_design_scores_gemma":[0.0002243914,0.001001738,0.1033077,0.002875944,0.0004922263,0.009174477,0.04337426,0.3494766,0.0391537,0.3314908,0.1187026,0.0007255227],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5603085,0.003442351,0.37811,0.03504883,0.0003611568,0.0003208539,0.0005784785,0.004975532,0.01685428],"genre_scores_gemma":[0.9415387,0.0003278274,0.05506808,0.001329783,0.00006433238,0.00006422903,0.0003108795,0.0007325776,0.0005635456],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.05005581,"threshold_uncertainty_score":0.2647236,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0263521555170315,"score_gpt":0.3083849450295481,"score_spread":0.2820327895125165,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}