{"id":"W7106806886","doi":"10.48448/0z2h-vg74","title":"Multilingual Collaborative Defense for Large Language Models","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Robustness (evolution); Safeguarding; Vulnerability (computing); Construct (python library); Language model; Code (set theory)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005684783,0.001942783,0.001294515,0.00124162,0.001266537,0.002368294,0.002906645,0.00195923,0.00512743],"category_scores_gemma":[0.01850398,0.0007065019,0.001597194,0.0007321659,0.001566244,0.004769655,0.006891143,0.0043407,0.003525932],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001451144,"about_ca_system_score_gemma":0.002218901,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004171902,"about_ca_topic_score_gemma":0.009212542,"domain_scores_codex":[0.9938498,0.002657954,0.0003293022,0.001626795,0.001015963,0.0005203056],"domain_scores_gemma":[0.9902132,0.00482109,0.0005423159,0.003178404,0.0008156456,0.0004292884],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009994599,0.0005236695,0.007233189,0.0006154981,0.0004414144,0.0005214526,0.0008106459,0.2875762,0.02014065,0.03202012,0.04236688,0.6067508],"study_design_scores_gemma":[0.00005183036,0.0001361786,0.0003014304,0.00002986051,0.00003044301,0.0001501786,0.0001476138,0.9565814,0.007422343,0.02946921,0.005648412,0.0000311791],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06469011,0.001595414,0.8873964,0.001754438,0.0002551073,0.000198199,0.001095412,0.03762529,0.005389583],"genre_scores_gemma":[0.6747414,0.000364005,0.3062176,0.001432113,0.0001865919,0.0003145912,0.005832918,0.003495754,0.007415066],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005684783,"threshold_uncertainty_score":0.03006434,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03825492641488465,"score_gpt":0.3723577619882639,"score_spread":0.3341028355733793,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}