{"id":"W4406866661","doi":"10.1145/3715109","title":"HumanEvalComm: Benchmarking the Communication Competence of Code Generation for LLMs and LLM Agent","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia, Okanagan Campus; Kelowna General Hospital; University of British Columbia","funders":"","keywords":"Benchmarking; Computer science; Competence (human resources); Knowledge management; Business; Psychology; Marketing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01098632,0.001110633,0.0005602631,0.001871208,0.0006846628,0.001875479,0.002246109,0.001942704,0.002431249],"category_scores_gemma":[0.05445615,0.0004311995,0.0006123267,0.001107209,0.001762966,0.002683057,0.003472951,0.00247168,0.001218786],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001540262,"about_ca_system_score_gemma":0.001925776,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004189781,"about_ca_topic_score_gemma":0.004271389,"domain_scores_codex":[0.9849145,0.009136591,0.001305703,0.001796796,0.002190084,0.0006563265],"domain_scores_gemma":[0.9365978,0.0412175,0.003197422,0.009975861,0.006299688,0.002711672],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007538795,0.01611206,0.09920377,0.004483792,0.00066604,0.001057969,0.01857329,0.1421435,0.04280531,0.01790292,0.04962459,0.5998879],"study_design_scores_gemma":[0.002244142,0.01376151,0.1157511,0.0005827784,0.0002504375,0.001070279,0.008640958,0.6559467,0.07521649,0.02047027,0.105457,0.0006084432],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9432372,0.0005036569,0.0356172,0.0005468635,0.0001624684,0.0009730293,0.001602592,0.006433391,0.01092362],"genre_scores_gemma":[0.9118089,0.0001599435,0.07518152,0.0003540929,0.00002981382,0.001255349,0.006818368,0.0006652083,0.003726848],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01098632,"threshold_uncertainty_score":0.05810189,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08995787232149585,"score_gpt":0.3486715527943801,"score_spread":0.2587136804728842,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}