{"id":"W7125019428","doi":"10.1109/aiware69974.2025.00018","title":"Software Testing with Large Language Models: An Interview Study with Practitioners","year":2025,"lang":"","type":"article","venue":"","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Test strategy; System integration testing; Process (computing); Qualitative research; Test (biology); Thematic analysis; Software; Personal software process; Manual testing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001417911,0.0004453825,0.0003981533,0.0002800598,0.0003445092,0.0009349392,0.001117763,0.00009853551,0.00006515475],"category_scores_gemma":[0.0002889002,0.0003288839,0.00004126921,0.001870198,0.00004673738,0.004046526,0.0004942269,0.000596261,0.000008559892],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008913987,"about_ca_system_score_gemma":0.0003566859,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006707112,"about_ca_topic_score_gemma":0.0005002227,"domain_scores_codex":[0.9973065,0.0003321587,0.0003947518,0.0009528184,0.0004659469,0.0005478182],"domain_scores_gemma":[0.9970946,0.000727958,0.0002550644,0.001379637,0.0003835875,0.0001591052],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005939298,0.01145439,0.04566655,0.001805886,0.001830196,0.002270009,0.04490319,0.05144051,0.00002694002,0.03876204,0.002321162,0.7989252],"study_design_scores_gemma":[0.002922086,0.01287744,0.00431961,0.003261773,0.0006217473,0.0002874801,0.01537169,0.9521898,0.000205829,0.001267443,0.004749712,0.001925361],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0138314,0.0006878462,0.9789539,0.0004974213,0.0001699942,0.0009714215,0.000006422123,0.002135568,0.002746004],"genre_scores_gemma":[0.5430244,0.00001818455,0.4557217,0.0004848529,0.00003012153,0.0000877217,0.000003640929,0.00002729904,0.0006020685],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9007493,"threshold_uncertainty_score":0.9999163,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04037925872012102,"score_gpt":0.3093107558454273,"score_spread":0.2689314971253063,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}