{"id":"W7109084896","doi":"10.1007/s10664-025-10768-1","title":"Output format biases in the evaluation of large language models for code translation","year":2025,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada; Vector Institute","keywords":"Executable; Source code; Disk formatting; Code (set theory); Code review; Translation (biology); Empirical research; Software","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03710896,0.001442283,0.001142588,0.002244888,0.00105627,0.004762754,0.001814593,0.002508873,0.003623045],"category_scores_gemma":[0.3115711,0.0007163321,0.0009893671,0.002467347,0.002073056,0.007129343,0.003557967,0.002905993,0.001227815],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002157331,"about_ca_system_score_gemma":0.002036998,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003691036,"about_ca_topic_score_gemma":0.003810913,"domain_scores_codex":[0.9609644,0.03073472,0.002299308,0.001934125,0.003616778,0.0004506866],"domain_scores_gemma":[0.5820019,0.3889491,0.005332486,0.01183287,0.01068428,0.001199555],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01144587,0.002084312,0.08483738,0.003337367,0.001491273,0.0008194621,0.007367271,0.3108533,0.01759399,0.03318355,0.01884673,0.5081394],"study_design_scores_gemma":[0.0009286639,0.001245471,0.01200035,0.0003844056,0.0005370032,0.0003363039,0.001536494,0.8998359,0.02881271,0.05004921,0.004175148,0.000158394],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8468252,0.001633839,0.1343043,0.00193273,0.0003332265,0.0003252197,0.001893814,0.004297646,0.008453984],"genre_scores_gemma":[0.9652106,0.000221313,0.02984457,0.0003320723,0.00008796525,0.0001628868,0.002304914,0.001003694,0.0008319087],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.962891,"threshold_uncertainty_score":0.1962533,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07234957541314808,"score_gpt":0.3707188625070804,"score_spread":0.2983692870939323,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}