{"id":"W4412945237","doi":"10.18653/v1/2025.acl-long.657","title":"Code-Switching Red-Teaming: LLM Evaluation for Safety and Multilingual Understanding","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Computer science; Code-switching; Code (set theory); Programming language; Software engineering; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004668144,0.001053851,0.0005396102,0.00110822,0.0006089906,0.001106963,0.001202424,0.001256839,0.002705335],"category_scores_gemma":[0.03025382,0.0002986789,0.0005790857,0.0006257485,0.001060968,0.002913585,0.002447715,0.001919753,0.001168853],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001107608,"about_ca_system_score_gemma":0.001192851,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004601903,"about_ca_topic_score_gemma":0.004991336,"domain_scores_codex":[0.9910676,0.004465698,0.0006713714,0.001105693,0.002307312,0.000382209],"domain_scores_gemma":[0.9778415,0.01397297,0.001247327,0.003200226,0.00301533,0.0007226413],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004293632,0.0035431,0.04411316,0.003741631,0.0003857806,0.001640704,0.02077824,0.1017884,0.1578542,0.01524404,0.05671817,0.5898989],"study_design_scores_gemma":[0.0004448373,0.002638743,0.01651333,0.0001937582,0.0001482083,0.00111015,0.007035004,0.8322582,0.09421954,0.006674444,0.03854213,0.0002217935],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8330894,0.0004843571,0.1288892,0.0006636017,0.0001628612,0.001081648,0.003124949,0.02210635,0.0103978],"genre_scores_gemma":[0.8871209,0.0001089491,0.1026631,0.0002734831,0.00003003638,0.0004323217,0.006005024,0.001279239,0.002086963],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004668144,"threshold_uncertainty_score":0.02468783,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08500175203487303,"score_gpt":0.383077898694948,"score_spread":0.298076146660075,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}