{"id":"W7124841212","doi":"10.1109/aiware69974.2025.00021","title":"CFCEval: Evaluating Security Aspects in Code Generated by Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Code (set theory); Metric (unit); Code review; Key (lock); Relevance (law)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008362384,0.001749172,0.0006293401,0.00369406,0.0005988063,0.002032551,0.002158997,0.001388031,0.001392834],"category_scores_gemma":[0.05145599,0.0004205932,0.001158228,0.002159027,0.001256458,0.003071234,0.001948628,0.0017576,0.000798773],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001794371,"about_ca_system_score_gemma":0.002719022,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007102378,"about_ca_topic_score_gemma":0.01230441,"domain_scores_codex":[0.990299,0.003929819,0.0008914551,0.001548825,0.003058335,0.0002726168],"domain_scores_gemma":[0.9571205,0.0296815,0.002385737,0.005098945,0.005057508,0.0006558028],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0022289,0.001266054,0.07100503,0.00432859,0.0009976886,0.00068358,0.001799757,0.272274,0.03525683,0.01164979,0.07854353,0.5199662],"study_design_scores_gemma":[0.0002850181,0.001270846,0.01476684,0.0002557209,0.000159557,0.0004400811,0.0003924377,0.9127145,0.03933311,0.01095339,0.01927542,0.0001529617],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6357321,0.005049748,0.2572672,0.001270234,0.0005541807,0.00116566,0.01987754,0.06965945,0.009423875],"genre_scores_gemma":[0.6873966,0.000807642,0.2548705,0.0005682764,0.00007646713,0.0008098246,0.04881204,0.00428925,0.002369353],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008362384,"threshold_uncertainty_score":0.04422504,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02988406004361717,"score_gpt":0.349527157522841,"score_spread":0.3196430974792238,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}