{"id":"W3138384652","doi":"10.1109/msr52588.2021.00035","title":"Leveraging Models to Reduce Test Cases in Software Repositories","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Test suite; Reduction (mathematics); Computer science; Fuzz testing; Test (biology); Test case; Compiler; Code coverage; Software; Reliability engineering; Data mining; Programming language; Machine learning; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00382561,0.002119468,0.001362798,0.003832955,0.0005665249,0.002488588,0.002954406,0.00167143,0.001732221],"category_scores_gemma":[0.03368139,0.001596477,0.00266334,0.001713354,0.001373379,0.003933395,0.00231762,0.002320437,0.0009628004],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001781708,"about_ca_system_score_gemma":0.003192037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007681364,"about_ca_topic_score_gemma":0.01156259,"domain_scores_codex":[0.9927972,0.002873084,0.0003825663,0.000905743,0.002641785,0.0003996562],"domain_scores_gemma":[0.9776145,0.01411194,0.001628832,0.004936994,0.001489568,0.0002180625],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003908647,0.0005348654,0.01275469,0.0004729202,0.00028121,0.00038078,0.0004471442,0.6731313,0.0297548,0.01024166,0.003998739,0.2676111],"study_design_scores_gemma":[0.00003369616,0.0001109913,0.0005795266,0.00002838497,0.00006815165,0.00009187132,0.00003443739,0.9772298,0.01183624,0.008538111,0.001424516,0.00002422299],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.129324,0.0005331831,0.8465059,0.0009095424,0.00005065613,0.0003913915,0.0004416402,0.0197698,0.002073701],"genre_scores_gemma":[0.5196378,0.0003610935,0.4729702,0.0003507443,0.00003976138,0.0005152137,0.002358049,0.002234822,0.001532277],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007681364,"threshold_uncertainty_score":0.02023196,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06181297279744596,"score_gpt":0.2997691108125746,"score_spread":0.2379561380151286,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}