{"id":"W7128191878","doi":"10.1145/3745764","title":"NLPerturbator: Studying the Robustness of Code LLMs to Natural Language Variations","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"National Key Research and Development Program of China","keywords":"Robustness (evolution); Natural language; Coding (social sciences); Code (set theory); Natural language generation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01346665,0.001466437,0.0008255481,0.002786577,0.0008196958,0.002069548,0.00227351,0.001579125,0.001218642],"category_scores_gemma":[0.1662024,0.0008184973,0.001006384,0.001661195,0.0027639,0.004728538,0.003065491,0.002569772,0.0007854066],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00193686,"about_ca_system_score_gemma":0.001893638,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00505538,"about_ca_topic_score_gemma":0.003392179,"domain_scores_codex":[0.9795815,0.009436825,0.001382843,0.003720497,0.005223567,0.0006547238],"domain_scores_gemma":[0.8282169,0.1202411,0.01527499,0.02738224,0.007645427,0.001239364],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003193958,0.001208417,0.08667507,0.002619471,0.0009984118,0.0007628776,0.004749591,0.4412093,0.08713088,0.01260527,0.008939668,0.349907],"study_design_scores_gemma":[0.0001494359,0.001557266,0.02301561,0.0001230845,0.0002295268,0.0005570541,0.0008071381,0.8969845,0.05319314,0.0159005,0.007291751,0.0001910046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.630388,0.001735231,0.3462412,0.0006959021,0.0001962826,0.0008765706,0.001489471,0.01426695,0.004110408],"genre_scores_gemma":[0.8875883,0.0003162656,0.1068349,0.0002893116,0.00005254454,0.000586302,0.002169569,0.001301538,0.0008612301],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9865333,"threshold_uncertainty_score":0.07121927,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05664408697152098,"score_gpt":0.3164152668179596,"score_spread":0.2597711798464386,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}