{"id":"W4310506204","doi":"10.1145/3563767.3568132","title":"Evaluating the Quality of Student-Written Software Tests with Curated Mutation Analysis","year":2022,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Test suite; Correctness; Programming language; Assertion; Software engineering; Software quality; Code coverage; Software; Software testing; Test (biology); Invocation; Test case; Software development; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008438538,0.0008531847,0.0006680313,0.002473543,0.0002617789,0.001624373,0.001187628,0.001274513,0.001885762],"category_scores_gemma":[0.1019311,0.0002589342,0.0005981477,0.001445324,0.0006351219,0.00115306,0.001185589,0.0007118813,0.0008521188],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006871691,"about_ca_system_score_gemma":0.0009029175,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002099542,"about_ca_topic_score_gemma":0.002785017,"domain_scores_codex":[0.9876993,0.003836259,0.0013918,0.001500375,0.005110151,0.0004620425],"domain_scores_gemma":[0.8765724,0.07201292,0.01268436,0.008971444,0.02763655,0.002122189],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006288883,0.003793196,0.2537588,0.001210617,0.001677648,0.001026438,0.002391779,0.14073,0.07871723,0.001578489,0.005785499,0.5030414],"study_design_scores_gemma":[0.0006730455,0.01227544,0.325602,0.0004623332,0.000650069,0.00166445,0.001080916,0.4753639,0.1695969,0.00296262,0.009329629,0.0003387273],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9448762,0.0004581345,0.04844362,0.0001322773,0.00008065459,0.0001563687,0.00060865,0.002447398,0.002796732],"genre_scores_gemma":[0.9593326,0.0001314361,0.03640898,0.0000782417,0.00001988118,0.00008191852,0.001719666,0.0004891802,0.001738055],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008438538,"threshold_uncertainty_score":0.04462779,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1151691530031579,"score_gpt":0.427437826411757,"score_spread":0.3122686734085992,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}