{"id":"W4398795735","doi":"10.1145/3664646.3665664","title":"AI-Assisted Assessment of Coding Practices in Modern Code Review","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"Google (Canada)","funders":"DeepMind","keywords":"Computer science; Java; Best practice; Software engineering; Python (programming language); Workflow; Coding (social sciences); Programming language; Software deployment; Code review; Static program analysis; Software development; Software; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03743833,0.0009344617,0.000966716,0.00676419,0.00166864,0.004486866,0.002798613,0.001621593,0.003802996],"category_scores_gemma":[0.1690612,0.0009143937,0.00064855,0.002552761,0.001386643,0.005554151,0.00424542,0.002968281,0.003115367],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002586505,"about_ca_system_score_gemma":0.007178938,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004941407,"about_ca_topic_score_gemma":0.01487052,"domain_scores_codex":[0.9540272,0.02699235,0.003133522,0.006396483,0.008662686,0.000787628],"domain_scores_gemma":[0.7380848,0.1560905,0.02080169,0.02955198,0.05065453,0.004816477],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0007080528,0.001080319,0.04571614,0.001378164,0.0002315494,0.0004739686,0.01108326,0.02217087,0.02931374,0.007752725,0.033588,0.8465032],"study_design_scores_gemma":[0.0005091456,0.0008719189,0.04010433,0.0007846829,0.0001552244,0.000824201,0.004412147,0.7553801,0.05842444,0.04259498,0.09544024,0.0004985912],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2333737,0.001340782,0.6819361,0.004134154,0.0005935454,0.00246601,0.001800185,0.0582909,0.01606457],"genre_scores_gemma":[0.415078,0.0003408152,0.5738813,0.0007671839,0.0001107576,0.0007532308,0.002120641,0.001518784,0.005429259],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9625617,"threshold_uncertainty_score":0.1979951,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1117525305005444,"score_gpt":0.4417333888704112,"score_spread":0.3299808583698668,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}