{"id":"W4403576996","doi":"10.48550/arxiv.2410.11769","title":"Can Search-Based Testing with Pareto Optimization Effectively Cover Failure-Revealing Test Inputs?","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission","keywords":"Cover (algebra); Pareto principle; Test (biology); Computer science; Pareto optimal; Reliability engineering; Mathematical optimization; Multi-objective optimization; Mathematics; Engineering; Geology; Paleontology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038754,0.001121587,0.001023831,0.001131002,0.0004252925,0.0009176395,0.001439769,0.00114019,0.001805676],"category_scores_gemma":[0.02169718,0.0003949477,0.0009776601,0.0008663908,0.001438117,0.002005059,0.001000407,0.001161343,0.0003516754],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001026314,"about_ca_system_score_gemma":0.002045584,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004728686,"about_ca_topic_score_gemma":0.004499441,"domain_scores_codex":[0.9978037,0.001018978,0.00008966099,0.0002368246,0.0005692944,0.0002814274],"domain_scores_gemma":[0.9911582,0.00650017,0.0005018819,0.0009266505,0.0007215557,0.0001915089],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001450418,0.0001414461,0.004207809,0.0001318751,0.00008057564,0.00009278605,0.0000847138,0.9022204,0.002844712,0.01112065,0.0009199567,0.07801003],"study_design_scores_gemma":[0.00002915548,0.0001097723,0.0005618464,0.00002812352,0.0000183801,0.00003087569,0.00004175485,0.98295,0.001581985,0.01411361,0.0005286569,0.000005921555],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1829373,0.0007288352,0.8071581,0.001103118,0.00005989279,0.0001423083,0.0001320577,0.00100849,0.006729948],"genre_scores_gemma":[0.853955,0.0002092362,0.144125,0.0002555314,0.00001866623,0.0001595236,0.0001762169,0.0001417986,0.0009590767],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004728686,"threshold_uncertainty_score":0.02049536,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05888171160340358,"score_gpt":0.1974152998554705,"score_spread":0.1385335882520669,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}