{"id":"W6979373727","doi":"","title":"LLM-Based Detection of Tangled Code Changes for Higher-Quality Method-Level Bug Datasets","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Winnipeg; University of Manitoba","funders":"","keywords":"Commit; Code (set theory); Granularity; Source code; Classifier (UML); Perceptron; Conflation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002286313,0.001545949,0.0007588746,0.002851632,0.0005462416,0.0008411085,0.001715158,0.001532288,0.001657431],"category_scores_gemma":[0.01509019,0.0003549655,0.00107088,0.001500112,0.0006013637,0.00226852,0.001658218,0.002188695,0.001584947],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00100221,"about_ca_system_score_gemma":0.001389222,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009301102,"about_ca_topic_score_gemma":0.01621459,"domain_scores_codex":[0.9981647,0.0003710867,0.0001675342,0.0007005685,0.000467959,0.0001281341],"domain_scores_gemma":[0.99303,0.003044843,0.0008098167,0.001469232,0.001312474,0.0003336139],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001512117,0.001779296,0.1360525,0.0016877,0.0004922433,0.001486401,0.001182043,0.1170563,0.03529296,0.002426025,0.1054585,0.5955739],"study_design_scores_gemma":[0.0001741245,0.0005568505,0.03021773,0.0001002934,0.0001097628,0.0004818316,0.0003430684,0.9239963,0.02675222,0.004224183,0.01293847,0.000105177],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7768014,0.003920213,0.10301,0.001447329,0.0005637927,0.0004142592,0.03093976,0.07977521,0.003128095],"genre_scores_gemma":[0.8143834,0.0004280117,0.1043327,0.0004493535,0.0001092673,0.0004231822,0.07472187,0.001287313,0.003864908],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009301102,"threshold_uncertainty_score":0.01849389,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1423767848433097,"score_gpt":0.3971744513948557,"score_spread":0.254797666551546,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}