{"id":"W1972275887","doi":"10.1109/saner.2015.7081831","title":"Detecting duplicate bug reports with software engineering domain knowledge","year":2015,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":39,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Data deduplication; Context (archaeology); Domain (mathematical analysis); Software; Word (group theory); Software bug; Software engineering; Information retrieval; Artificial intelligence; Database; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000774747,0.0001784836,0.0001641259,0.0001897149,0.00005821297,0.0001747023,0.0005591877,0.00005885592,0.000005345863],"category_scores_gemma":[0.001359923,0.0001473228,0.00003441185,0.0007996675,0.00001923719,0.0003751472,0.0003983423,0.0002409781,0.00007894621],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001417071,"about_ca_system_score_gemma":0.000167567,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002950383,"about_ca_topic_score_gemma":0.000008082975,"domain_scores_codex":[0.9983949,0.00001866073,0.0002108137,0.0004903885,0.0003945122,0.0004907374],"domain_scores_gemma":[0.9980921,0.0004174788,0.00004395356,0.0009000172,0.0002066272,0.0003397812],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008822235,0.0007300078,0.6237608,0.0008608684,0.0005671987,0.01002517,0.01830575,0.1494181,0.009953352,0.01994112,0.01876003,0.1475893],"study_design_scores_gemma":[0.006534229,0.002769515,0.1674899,0.001149985,0.00005676514,0.01882816,0.0006185917,0.4512128,0.133668,0.006190734,0.2032447,0.008236681],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1500873,0.0001492946,0.8470984,0.00008501438,0.0002531671,0.0001433205,1.364609e-7,0.001815491,0.0003679064],"genre_scores_gemma":[0.6540632,4.904911e-7,0.3454286,0.00001267434,0.00007079945,0.00003866764,5.070612e-7,0.00002734352,0.0003578257],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.5039759,"threshold_uncertainty_score":0.600765,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02091294341494524,"score_gpt":0.2482772196194589,"score_spread":0.2273642762045136,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}