{"id":"W2011861933","doi":"10.1109/csmr.2012.78","title":"A Comparative Study of the Performance of IR Models on Duplicate Bug Detection","year":2012,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Heuristics; Weighting; Set (abstract data type); Data mining; Entropy (arrow of time); Machine learning; Artificial intelligence; Information retrieval; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000223836,0.00005186674,0.00009317236,0.00005429912,0.00003398301,0.000006570476,0.0004607013,0.00001446371,0.000002081736],"category_scores_gemma":[0.00001827321,0.00003229815,0.00001988464,0.0003502757,0.00001975257,0.0002318161,0.0001639934,0.0000882475,0.000007620303],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002154822,"about_ca_system_score_gemma":0.00001033524,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003809999,"about_ca_topic_score_gemma":0.000004138457,"domain_scores_codex":[0.9993518,0.00003424958,0.0001073688,0.00009505101,0.0002807058,0.0001308085],"domain_scores_gemma":[0.9992662,0.0001642172,0.00003830163,0.0004384399,0.00006508923,0.00002772117],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001081637,0.00393405,0.3029871,0.0001401875,0.0001749566,3.706527e-7,0.04915687,0.580054,0.03884275,0.00798146,0.0002180719,0.01640196],"study_design_scores_gemma":[0.0001807726,0.0003927734,0.3187278,0.000009109003,0.000001831515,0.000001029977,0.00009277725,0.507646,0.1728664,0.00002528233,0.000006501618,0.00004972637],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9303915,0.00000923236,0.06900127,0.00001115176,0.00008595635,0.0002143141,1.577776e-7,0.0000416484,0.0002447659],"genre_scores_gemma":[0.9993313,7.171665e-7,0.0005628468,0.00000449015,0.000009929643,0.00001892023,1.998298e-8,0.000002561837,0.00006923355],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1340236,"threshold_uncertainty_score":0.1317081,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05731752959737332,"score_gpt":0.2968354962486602,"score_spread":0.2395179666512869,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}