{"id":"W1970652823","doi":"10.1145/1808901.1808908","title":"Finding similar defects using synonymous identifier retrieval","year":2010,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; University of Waterloo","keywords":"Identifier; Fragment (logic); Computer science; Code (set theory); Source code; clone (Java method); Data mining; Information retrieval; Programming language; Biology; Set (abstract data type); Gene; Genetics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001544113,0.0007911583,0.001722787,0.01221135,0.001133752,0.001736042,0.00208736,0.00192949,0.002123098],"category_scores_gemma":[0.01068479,0.0003879858,0.001169479,0.006343527,0.0008399448,0.004128122,0.002756065,0.0006547624,0.001279919],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007410937,"about_ca_system_score_gemma":0.001420795,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003113103,"about_ca_topic_score_gemma":0.003413648,"domain_scores_codex":[0.9973344,0.0003994872,0.0003212932,0.0006313663,0.001136157,0.0001774351],"domain_scores_gemma":[0.9943356,0.001893025,0.0009071456,0.001116915,0.001545891,0.0002014106],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006109508,0.0003356612,0.03705673,0.00104145,0.0003140598,0.002912674,0.002125535,0.004797725,0.1702861,0.01125803,0.01114429,0.7581168],"study_design_scores_gemma":[0.0004800533,0.001898831,0.08137543,0.0003478757,0.001588285,0.0277526,0.004086996,0.4332794,0.3195758,0.04356582,0.0852891,0.0007598753],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.270941,0.002323836,0.7059316,0.0006098515,0.0002131934,0.0005211523,0.001513879,0.01180588,0.006139665],"genre_scores_gemma":[0.442257,0.0009382862,0.546343,0.0003741357,0.0001742614,0.0002717622,0.005101384,0.0007044799,0.003835791],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01221135,"threshold_uncertainty_score":0.008166194,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02937027681914007,"score_gpt":0.2973919301682039,"score_spread":0.2680216533490638,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}