{"id":"W4403604523","doi":"10.1145/3672448","title":"Understanding Test Convention Consistency as a Dimension of Test Quality","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test (biology); Consistency (knowledge bases); Dimension (graph theory); Quality (philosophy); Reliability engineering; Artificial intelligence; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.001384237,0.0001849205,0.0002994296,0.0004612432,0.00009240827,0.00005794499,0.0003118032,0.000150432,0.00001905924],"category_scores_gemma":[0.0102506,0.000185409,0.0001063456,0.0006088142,0.00007915039,0.0001989706,0.00003179273,0.0004016448,0.0000128876],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001117697,"about_ca_system_score_gemma":0.0000865195,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004136064,"about_ca_topic_score_gemma":0.000002194035,"domain_scores_codex":[0.9985633,0.0001360625,0.0003287341,0.0004476871,0.0002342145,0.0002899724],"domain_scores_gemma":[0.9702404,0.02896531,0.00003516346,0.0005734338,0.00006495762,0.0001207581],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001616218,0.001399223,0.01717038,0.008573256,0.001450172,0.0005657714,0.01022121,0.08911382,0.2795537,0.2677137,0.0004980623,0.323579],"study_design_scores_gemma":[0.007895525,0.01140254,0.113908,0.007780372,0.000907608,0.004858965,0.001792051,0.3745534,0.362061,0.09427913,0.01351977,0.007041699],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01544165,0.0006391564,0.9817725,0.0002717464,0.0008199531,0.0001326347,0.00001393251,0.0008986735,0.000009782948],"genre_scores_gemma":[0.6310542,0.0001049623,0.3686546,0.00001924662,0.00001791204,0.00002120752,0.000001720115,0.00002103899,0.0001050638],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.6156126,"threshold_uncertainty_score":0.9980865,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2278846789678397,"score_gpt":0.37761842431744,"score_spread":0.1497337453496004,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}