{"id":"W4403780126","doi":"10.48550/arxiv.2409.14610","title":"An Empirical Study of Refactoring Engine Bugs","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Hydraulic and Pneumatic Systems","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code refactoring; Computer science; Empirical research; Programming language; Software engineering; Software; Mathematics; Statistics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008359869,0.0004771288,0.0003775697,0.004172495,0.0006565587,0.0008988853,0.0009636474,0.0009571408,0.001206247],"category_scores_gemma":[0.1013926,0.0003689632,0.0004350282,0.00290121,0.00128946,0.002544115,0.00119937,0.001082337,0.0003316503],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007567476,"about_ca_system_score_gemma":0.0007426314,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001924446,"about_ca_topic_score_gemma":0.002184072,"domain_scores_codex":[0.9896842,0.003024185,0.001781792,0.001317465,0.003648693,0.0005436658],"domain_scores_gemma":[0.7833378,0.1334708,0.04980625,0.007104581,0.02364608,0.002634579],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001668351,0.0004617009,0.9662281,0.0002631078,0.00007199582,0.000571656,0.005706337,0.0002946886,0.001173727,0.0003439283,0.0006960784,0.02402178],"study_design_scores_gemma":[0.00003079031,0.0007450765,0.9818308,0.0001698374,0.00007312129,0.001684662,0.007010486,0.003084891,0.001781824,0.0003953119,0.003144937,0.0000484261],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9978068,0.0002392443,0.0008873585,0.00009678879,0.000006403681,0.00005999211,0.0002780056,0.00002532048,0.0006001388],"genre_scores_gemma":[0.9976864,0.0001700626,0.001255178,0.0000550587,0.000007091254,0.00007155427,0.0004700047,0.00002234692,0.0002623694],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008359869,"threshold_uncertainty_score":0.04421175,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08149237320390153,"score_gpt":0.2191156387319377,"score_spread":0.1376232655280361,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}