{"id":"W7109084896","doi":"10.1007/s10664-025-10768-1","title":"Output format biases in the evaluation of large language models for code translation","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Vector Institute","keywords":"Executable; Source code; Disk formatting; Code (set theory); Code review; Translation (biology); Empirical research; Software","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001109852,0.00008374727,0.0001120948,0.000147175,0.00003221777,0.00003803697,0.0004197182,0.00006051171,7.829e-7],"category_scores_gemma":[0.0007705418,0.00006258894,0.00005051588,0.0004524664,0.000005792254,0.0003522066,0.0000416138,0.0001050661,2.16785e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006446775,"about_ca_system_score_gemma":0.00005478014,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000007076374,"about_ca_topic_score_gemma":0.00000938769,"domain_scores_codex":[0.9991794,0.00003932045,0.0001968701,0.0001437215,0.0002677015,0.0001729668],"domain_scores_gemma":[0.9991007,0.0005500224,0.00003553483,0.0002145579,0.00008527847,0.00001386392],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004152625,0.0003501737,0.002786603,0.0009626055,0.00005932725,0.000007367986,0.03760976,0.3793203,0.001439257,0.02281322,0.003272653,0.5513372],"study_design_scores_gemma":[0.000303485,0.00001544021,0.0004704239,0.0001132704,0.00001143052,9.926662e-7,0.00003269031,0.9889132,0.003216639,0.00651692,0.0003296335,0.00007587794],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02565829,0.003129102,0.9701245,0.0004002058,0.00006383866,0.0003453087,0.000009061304,0.0002517155,0.00001800524],"genre_scores_gemma":[0.7556815,0.000002204044,0.244063,0.0001530739,0.00001148043,0.00007129578,0.000008725264,0.000004282391,0.000004477499],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7300232,"threshold_uncertainty_score":0.2552303,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07234957541314808,"score_gpt":0.3707188625070804,"score_spread":0.2983692870939323,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}