{"id":"W4308643083","doi":"10.1145/3540250.3558952","title":"Leveraging test plan quality to improve code review efficacy","year":2022,"lang":"en","type":"article","venue":"Proceedings of the 30th ACM Joint European Software Engineering Conference and Symposium on the Foundations of Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Code review; Leverage (statistics); Documentation; Code coverage; Test case; Source code; Natural language; Transformer; Test (biology); Software engineering; Software quality; Artificial intelligence; Data science; Information retrieval; Natural language processing; Machine learning; Programming language; Software; Software development; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03730888,0.00106052,0.001141665,0.005231675,0.0005251719,0.00295336,0.001315919,0.001241012,0.00112646],"category_scores_gemma":[0.1925338,0.0004380243,0.0009326666,0.002057975,0.0008385202,0.004814939,0.001448419,0.00152101,0.0007522406],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001837968,"about_ca_system_score_gemma":0.002443057,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004338627,"about_ca_topic_score_gemma":0.006764498,"domain_scores_codex":[0.9726774,0.01347584,0.002668897,0.004182758,0.006366058,0.0006292044],"domain_scores_gemma":[0.715607,0.227215,0.02209639,0.008576124,0.02449212,0.002013299],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00175772,0.001197171,0.3950898,0.001653754,0.00100555,0.0003291106,0.003772414,0.06458274,0.01551488,0.002140743,0.008379116,0.5045771],"study_design_scores_gemma":[0.0001505102,0.00141843,0.107098,0.0001559336,0.00045299,0.0003206968,0.0006519571,0.864974,0.01452255,0.006126743,0.003984049,0.0001441024],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7216461,0.002895004,0.2609076,0.002031605,0.0001951699,0.00105075,0.001171844,0.005180634,0.004921151],"genre_scores_gemma":[0.9721437,0.0001950312,0.02560011,0.0001746827,0.00005349168,0.0001234555,0.0009709038,0.000134182,0.0006044214],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03730888,"threshold_uncertainty_score":0.1973106,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04372462117162971,"score_gpt":0.2599137776609984,"score_spread":0.2161891564893687,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}