{"id":"W4391136370","doi":"10.1145/3641541","title":"Learning Failure-Inducing Models for Testing Software-Defined Networks","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software-Defined Networks and 5G","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg; Science Foundation Ireland","keywords":"Computer science; Software testing; Software engineering; Software; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001277059,0.0003777732,0.0004527131,0.0003881429,0.0003803593,0.0002388309,0.0005205642,0.0003170196,0.000004961932],"category_scores_gemma":[0.002584473,0.000376519,0.0001772804,0.0008154567,0.00003810359,0.0003704324,0.00004686113,0.0008677286,0.000004444103],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006169785,"about_ca_system_score_gemma":0.00007399729,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002955147,"about_ca_topic_score_gemma":0.000004389921,"domain_scores_codex":[0.997741,0.0001677811,0.000384543,0.0008628854,0.00015669,0.0006870483],"domain_scores_gemma":[0.9842752,0.01481706,0.00005491091,0.0005734317,0.0000992028,0.0001802201],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000008492663,0.000009413383,0.00002527822,0.00008297127,0.00006655804,0.0000102255,0.0002846677,0.6563078,0.00005991029,0.002720535,0.00004319015,0.340381],"study_design_scores_gemma":[0.0003428526,0.0003568791,0.00006077998,0.0002619939,0.00007398643,0.0001854187,0.00002709847,0.9845549,0.0001947865,0.009494428,0.003968606,0.0004782723],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001314266,0.001662771,0.9913774,0.0002741679,0.001703401,0.0002561183,0.000006327202,0.0034012,0.000004318486],"genre_scores_gemma":[0.1048759,0.0001018567,0.8942987,0.0001339932,0.0002179768,0.0001783,0.000006247811,0.00007994899,0.0001070288],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.3399028,"threshold_uncertainty_score":0.9998687,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08763148271628124,"score_gpt":0.29222353392917,"score_spread":0.2045920512128888,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}