{"id":"W7117477189","doi":"10.1145/3786771","title":"Assessing and Advancing Benchmarks for Evaluating Large Language Models in Software Engineering Tasks","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Popularity; Software quality assurance; Model-driven architecture; Software development; Coding (social sciences); Quality (philosophy); Software; Benchmark (surveying); Social software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02847464,0.002102741,0.001180832,0.007093144,0.00109781,0.004469833,0.003321755,0.001570603,0.002109651],"category_scores_gemma":[0.1509183,0.000640678,0.001008586,0.007185057,0.001262063,0.00547589,0.003611095,0.002466268,0.000944953],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003114455,"about_ca_system_score_gemma":0.004158453,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006714359,"about_ca_topic_score_gemma":0.009007523,"domain_scores_codex":[0.9612986,0.02100643,0.004160275,0.001795006,0.01061975,0.001119984],"domain_scores_gemma":[0.8495432,0.100462,0.008795676,0.0143035,0.02459909,0.002296632],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001515574,0.002197429,0.04428716,0.005456994,0.0005786169,0.0003161597,0.00223377,0.2161458,0.01531095,0.05814164,0.03178678,0.6220291],"study_design_scores_gemma":[0.0003695996,0.002782533,0.02373221,0.002703756,0.0003316777,0.0004291721,0.002290535,0.8092759,0.04039809,0.06564186,0.05172943,0.0003151341],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3523613,0.01732902,0.5757956,0.002947687,0.001017999,0.002164639,0.005494248,0.01187749,0.03101199],"genre_scores_gemma":[0.5021333,0.003706838,0.4746594,0.0004448142,0.0001590699,0.001968819,0.01294792,0.001890595,0.002089364],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9715254,"threshold_uncertainty_score":0.1505901,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07511031672640531,"score_gpt":0.3920066529526886,"score_spread":0.3168963362262833,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}