{"id":"W2037556552","doi":"10.5555/776816.776826","title":"Using benchmarking to advance research: a challenge to software engineering","year":2003,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":200,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; University of Toronto","funders":"","keywords":"Benchmarking; Benchmark (surveying); Computer science; Data science; Software; Software engineering; Academic community; Management science; Engineering management; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2759394,0.002974036,0.005045649,0.01145303,0.00555201,0.02779631,0.01021218,0.008602041,0.001746849],"category_scores_gemma":[0.5178069,0.00129715,0.001566891,0.01781468,0.01990489,0.0545129,0.01899195,0.01585046,0.002193484],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006429577,"about_ca_system_score_gemma":0.02227413,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00272229,"about_ca_topic_score_gemma":0.002457725,"domain_scores_codex":[0.6318311,0.2453572,0.02013659,0.01417041,0.08441986,0.004084905],"domain_scores_gemma":[0.3023207,0.4588483,0.02682273,0.1113942,0.08751724,0.013097],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000115957,0.0003181491,0.005439621,0.001757482,0.0001663121,0.0001655281,0.003754103,0.01088346,0.001181667,0.4884381,0.03394277,0.4538368],"study_design_scores_gemma":[0.00007579211,0.0003725815,0.001416887,0.001985407,0.00004561293,0.0001585877,0.002867356,0.01585282,0.001502985,0.8726366,0.1028949,0.0001904917],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02034564,0.02646788,0.6693411,0.2460971,0.003796527,0.0005994729,0.0003300469,0.003449852,0.02957244],"genre_scores_gemma":[0.2186676,0.01460579,0.7433994,0.0126248,0.003080432,0.001923728,0.0007378554,0.002158564,0.002801745],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7240605,"threshold_uncertainty_score":0.8928956,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.112132774218325,"score_gpt":0.3623551205264038,"score_spread":0.2502223463080788,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}