{"id":"W4410344404","doi":"10.1145/3735553","title":"Large Language Models for Automated Web-Form-Test Generation: An Empirical Study","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Test (biology); Software engineering; Web application; Natural language processing; Programming language; World Wide Web; Information retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02084597,0.001419794,0.0009199445,0.001822605,0.0007405793,0.002551958,0.002013888,0.001418776,0.002690488],"category_scores_gemma":[0.1549455,0.0007902036,0.00135645,0.001959577,0.001377765,0.005385495,0.001946805,0.003209812,0.001251025],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002311879,"about_ca_system_score_gemma":0.002141302,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009740454,"about_ca_topic_score_gemma":0.007872649,"domain_scores_codex":[0.9797032,0.01367269,0.001501491,0.001771003,0.002876724,0.0004748492],"domain_scores_gemma":[0.6626403,0.306694,0.007829366,0.01215211,0.008890552,0.001793648],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004241603,0.01230261,0.2810479,0.003550728,0.0006507839,0.001401296,0.007422355,0.1570182,0.007697104,0.004962047,0.01691025,0.5027951],"study_design_scores_gemma":[0.0005258867,0.00210429,0.06903405,0.0003881955,0.000517313,0.0006827799,0.002826927,0.9057243,0.007264513,0.003077704,0.007693443,0.0001605278],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9754132,0.0009413905,0.01862554,0.0003968624,0.0000426841,0.0005138186,0.001053768,0.001078884,0.001933892],"genre_scores_gemma":[0.9663121,0.0003511633,0.02857573,0.0001617836,0.00003496844,0.0004462553,0.003326062,0.0002430731,0.0005489484],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02084597,"threshold_uncertainty_score":0.1102453,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1082660934781946,"score_gpt":0.3902024629079375,"score_spread":0.281936369429743,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}