{"id":"W4391974543","doi":"10.1109/tse.2024.3368208","title":"Software Testing With Large Language Models: Survey, Landscape, and Vision","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":394,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Software engineering; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009385707,0.001491541,0.001848627,0.00599255,0.0004309003,0.004482325,0.003635035,0.002141713,0.002385037],"category_scores_gemma":[0.06181108,0.00100749,0.001519923,0.005038078,0.003055187,0.01052341,0.002492735,0.003093329,0.001103475],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002108319,"about_ca_system_score_gemma":0.00291645,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006331891,"about_ca_topic_score_gemma":0.004358084,"domain_scores_codex":[0.9900885,0.004911423,0.0006522069,0.001158091,0.002961443,0.0002282895],"domain_scores_gemma":[0.9138474,0.07639063,0.002021867,0.002982831,0.004228889,0.0005283364],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000145883,0.0002058172,0.01046892,0.003307177,0.0001709392,0.0001961956,0.0005721956,0.01650292,0.001008697,0.02595612,0.01321999,0.9282451],"study_design_scores_gemma":[0.0001181542,0.0008701161,0.01419034,0.00932717,0.0004632761,0.002972635,0.002998374,0.4816467,0.008588703,0.1844773,0.2940071,0.0003400746],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.03519735,0.4594508,0.4426527,0.03379083,0.0006956821,0.0003308217,0.000978464,0.00620898,0.02069431],"genre_scores_gemma":[0.4255341,0.3314767,0.2209292,0.008213248,0.003103178,0.0005827866,0.003668368,0.001898819,0.004593555],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.009385707,"threshold_uncertainty_score":0.04963696,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01712040667441351,"score_gpt":0.2519216021540697,"score_spread":0.2348011954796562,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}