{"id":"W4205513494","doi":"10.1109/ase51524.2021.9678640","title":"Is Historical Data an Appropriate Benchmark for Reviewer Recommendation Systems? : A Case Study of the Gerrit Community","year":2021,"lang":"en","type":"article","venue":"2021 36th IEEE/ACM International Conference on Automated Software Engineering (ASE)","topic":"Software Engineering Research","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; Polytechnique Montréal; McGill University","funders":"","keywords":"Context (archaeology); Pessimism; Computer science; Benchmark (surveying); Replicate; Task (project management); Data science; History; Epistemology; Management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03990285,0.0004654721,0.0007826616,0.004814048,0.003822972,0.003339028,0.001904087,0.002095913,0.001294335],"category_scores_gemma":[0.1941079,0.0004792735,0.0005547709,0.004891448,0.001476458,0.004909103,0.001803669,0.001491019,0.0007009175],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002961204,"about_ca_system_score_gemma":0.002903485,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02088401,"about_ca_topic_score_gemma":0.04176582,"domain_scores_codex":[0.9538686,0.030837,0.002590176,0.003709339,0.008021301,0.0009735426],"domain_scores_gemma":[0.6025497,0.2783934,0.02805292,0.02027108,0.06323623,0.007496602],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001634384,0.001946606,0.4526021,0.002860971,0.0003622911,0.007178135,0.1841349,0.00649564,0.009586056,0.003111941,0.01575354,0.3143335],"study_design_scores_gemma":[0.0005145927,0.005599404,0.5615504,0.002055454,0.0004527682,0.007123231,0.2091733,0.06415552,0.0165475,0.005892315,0.1260109,0.0009246325],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9839527,0.001148049,0.007267767,0.001804078,0.00005674351,0.0004369847,0.0005413618,0.0002659285,0.004526332],"genre_scores_gemma":[0.9826123,0.0003752549,0.01402393,0.0003155615,0.00007086336,0.0002358108,0.0006293292,0.000125736,0.001611189],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9600971,"threshold_uncertainty_score":0.211029,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1459378586426111,"score_gpt":0.3631536798642139,"score_spread":0.2172158212216028,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}