{"id":"W7124996121","doi":"10.1109/aiware69974.2025.00012","title":"Turning Manual Tasks Into Actions: Assessing the Effectiveness of Gemini-Generated Selenium Tests","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Executable; HTML; Software; Test (biology); Hypertext; User interface","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008093501,0.001355653,0.0006403725,0.002197915,0.0003250708,0.001285178,0.001452976,0.0009662763,0.001326579],"category_scores_gemma":[0.08970255,0.0004636096,0.0004800141,0.0009295448,0.000829943,0.001152191,0.001322028,0.0008166509,0.0008152339],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006981266,"about_ca_system_score_gemma":0.0006182345,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002471359,"about_ca_topic_score_gemma":0.002368445,"domain_scores_codex":[0.9911788,0.003474722,0.001089437,0.001211578,0.00264933,0.0003961226],"domain_scores_gemma":[0.8794248,0.09667136,0.007594491,0.008544275,0.006432122,0.001333015],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.009363849,0.0050608,0.1398229,0.002194406,0.0005894634,0.0008915826,0.005135187,0.1296097,0.08985608,0.00189034,0.007081907,0.6085038],"study_design_scores_gemma":[0.0008983638,0.01769195,0.1849518,0.0003962209,0.0005664486,0.001127495,0.002617488,0.5333204,0.2391896,0.002275341,0.01656953,0.000395317],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9760572,0.0003972346,0.01588865,0.00009796173,0.0000461076,0.000256967,0.0004609769,0.004017249,0.002777789],"genre_scores_gemma":[0.9539682,0.0001518861,0.0409544,0.0001164929,0.00001824055,0.0002258528,0.002311771,0.0008409109,0.001412256],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008093501,"threshold_uncertainty_score":0.04280305,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02708333145676899,"score_gpt":0.3548974381655703,"score_spread":0.3278141067088013,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}