{"id":"W4394015350","doi":"10.1016/j.infsof.2024.107468","title":"Effective test generation using pre-trained Large Language Models and mutation testing","year":2024,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":97,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs; Canadian Institute for Advanced Research","keywords":"Test (biology); Mutation; Computer science; Natural language processing; Genetics; Biology; Gene","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001237452,0.001579311,0.001016352,0.00133628,0.0003595494,0.0007329915,0.001884496,0.00151937,0.001727021],"category_scores_gemma":[0.008385114,0.0004827386,0.001178309,0.0006041561,0.0006916634,0.001281387,0.001002052,0.001509285,0.0008576497],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008024166,"about_ca_system_score_gemma":0.001382725,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004964372,"about_ca_topic_score_gemma":0.006390506,"domain_scores_codex":[0.9986789,0.0004145021,0.0000696754,0.0004246544,0.0002971177,0.0001151506],"domain_scores_gemma":[0.9941551,0.004301644,0.0003445081,0.0004481661,0.0005950739,0.0001554849],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00046,0.0006600637,0.008439554,0.0004469495,0.0001737007,0.001282466,0.000273793,0.3925345,0.03624671,0.0020628,0.008417867,0.5490016],"study_design_scores_gemma":[0.00003307246,0.00009789668,0.0005333508,0.00001355154,0.0000236734,0.0001596766,0.00002871656,0.9910425,0.005875171,0.001592714,0.0005876768,0.00001200032],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1823681,0.0009965552,0.7857834,0.000658948,0.0001264453,0.0003131207,0.000635537,0.02716559,0.001952246],"genre_scores_gemma":[0.7050749,0.0001941245,0.2882021,0.0007379483,0.00006047526,0.0003441855,0.002924585,0.0006687777,0.001792911],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004964372,"threshold_uncertainty_score":0.009870946,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01684748034253352,"score_gpt":0.2775876674362418,"score_spread":0.2607401870937083,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}