{"id":"W4412703957","doi":"10.1145/3696630.3728608","title":"A Tool for Generating Exceptional Behavior Tests With Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Alliance de recherche numérique du Canada; University of Texas at Austin; Cisco Systems; National Science Foundation","keywords":"Computer science; Programming language; Artificial intelligence; Natural language processing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003573802,0.002428494,0.000705377,0.002274638,0.0007511177,0.002183702,0.00268206,0.001712846,0.02036372],"category_scores_gemma":[0.02079495,0.002148898,0.002026309,0.0008745727,0.001270652,0.00449354,0.003992781,0.002959709,0.006954796],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009144454,"about_ca_system_score_gemma":0.002081499,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002517139,"about_ca_topic_score_gemma":0.004676433,"domain_scores_codex":[0.9975484,0.0007269623,0.0002848985,0.0004355437,0.0008552209,0.0001488791],"domain_scores_gemma":[0.9885709,0.007940791,0.0006624224,0.001686156,0.0009182722,0.0002215333],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001023841,0.0009577479,0.01909916,0.003190887,0.0004422535,0.003341195,0.0032406,0.09731524,0.05694107,0.09439126,0.2568735,0.4631833],"study_design_scores_gemma":[0.0004004732,0.0002607252,0.001798643,0.0005169809,0.0001178296,0.001590702,0.0003434123,0.6649532,0.05995357,0.06744041,0.2024116,0.0002124455],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007099352,0.0001500794,0.7154246,0.0003119792,0.0000941939,0.0003313469,0.004467019,0.2679206,0.004200891],"genre_scores_gemma":[0.124522,0.0003257328,0.7956105,0.0004702585,0.00006350029,0.001278164,0.0180384,0.05206231,0.007629149],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02036372,"threshold_uncertainty_score":0.0681234,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03533501993292639,"score_gpt":0.3178231966061725,"score_spread":0.2824881766732462,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}