{"id":"W2757935660","doi":"10.1109/tse.2017.2755005","title":"Revisiting the Performance Evaluation of Automated Approaches for the Retrieval of Duplicate Issue Reports","year":2017,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Engineering Research","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Information retrieval; Eclipse; Categorical variable; Software; Notation; Data mining; Machine learning; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02595014,0.002935118,0.002799891,0.009422928,0.001780176,0.004599285,0.005174393,0.002700436,0.001995583],"category_scores_gemma":[0.1005504,0.0007586177,0.001642233,0.006633272,0.001803985,0.007311249,0.002460938,0.002390605,0.002507031],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003279811,"about_ca_system_score_gemma":0.005136827,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03327981,"about_ca_topic_score_gemma":0.0256014,"domain_scores_codex":[0.9527919,0.01505496,0.005800851,0.006950428,0.01757103,0.001830739],"domain_scores_gemma":[0.8903194,0.0606559,0.006545838,0.01857572,0.02167656,0.002226623],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003186849,0.00274815,0.03501901,0.00544623,0.001670456,0.0005419428,0.001464802,0.05342619,0.02561179,0.002721983,0.07453018,0.7936324],"study_design_scores_gemma":[0.001141549,0.005628568,0.07657997,0.0008575051,0.001216223,0.002139449,0.00253084,0.7455607,0.07486661,0.004372637,0.08444674,0.0006591755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7607508,0.03915245,0.1125188,0.002439656,0.002219144,0.00203708,0.009971064,0.05404164,0.01686936],"genre_scores_gemma":[0.7633595,0.003339672,0.1995482,0.0007670198,0.0005705563,0.0005544062,0.0261067,0.001418017,0.004335886],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03327981,"threshold_uncertainty_score":0.1372392,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05980927837122244,"score_gpt":0.2995360331029995,"score_spread":0.2397267547317771,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}