{"id":"W4406866661","doi":"10.1145/3715109","title":"HumanEvalComm: Benchmarking the Communication Competence of Code Generation for LLMs and LLM Agent","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia, Okanagan Campus; Kelowna General Hospital; University of British Columbia","funders":"","keywords":"Benchmarking; Computer science; Competence (human resources); Knowledge management; Business; Psychology; Marketing","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007843094,0.00008422558,0.000137979,0.0001217946,0.0001889635,0.0000314615,0.0003922682,0.00006880033,8.259618e-7],"category_scores_gemma":[0.0004165534,0.00006811858,0.00002885455,0.0001604944,0.00005409734,0.00008905956,0.00003299369,0.0001584151,4.215692e-8],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001571663,"about_ca_system_score_gemma":0.00001662507,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002116386,"about_ca_topic_score_gemma":0.00001276496,"domain_scores_codex":[0.9993891,0.0001329809,0.0001550176,0.0001750207,0.0000505914,0.00009733547],"domain_scores_gemma":[0.997439,0.001976122,0.0000484971,0.0004551998,0.00006424663,0.00001692873],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000193266,0.00005016173,0.00006484689,0.0002533664,0.00008234118,5.504906e-7,0.001926167,0.006092675,0.0219586,0.06686617,0.00008747423,0.9025983],"study_design_scores_gemma":[0.001371154,0.0007729082,0.001828833,0.000698606,0.0002569952,0.0001010922,0.0002068116,0.6039239,0.30344,0.08010735,0.006527245,0.0007652055],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004324777,0.002834901,0.9917193,0.0006293422,0.0001675145,0.0001584297,0.000004046915,0.0001599774,0.000001699422],"genre_scores_gemma":[0.1873408,0.0001940071,0.8122547,0.0001272584,0.000008655018,0.00004855698,0.000002451023,0.000004046498,0.00001957716],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9018331,"threshold_uncertainty_score":0.2777795,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08995787232149585,"score_gpt":0.3486715527943801,"score_spread":0.2587136804728842,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}