{"id":"W4409348661","doi":"10.1609/aaai.v39i27.35029","title":"J&amp;H: Evaluating the Robustness of Large Language Models Under Knowledge-Injection Attacks in Legal Domain","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Robustness (evolution); Computer science; Computer security; Chemistry; Biochemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0262284,0.001694125,0.000952852,0.003118367,0.001146326,0.002817942,0.002827026,0.003477872,0.001834212],"category_scores_gemma":[0.1308022,0.0007218606,0.001712773,0.001260636,0.003183741,0.007326817,0.004006691,0.003698647,0.0007757114],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002116651,"about_ca_system_score_gemma":0.002750652,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01501726,"about_ca_topic_score_gemma":0.009502847,"domain_scores_codex":[0.9726312,0.01356876,0.002607062,0.00381951,0.00624784,0.001125711],"domain_scores_gemma":[0.8271473,0.1352598,0.009048958,0.0198337,0.005693456,0.003016753],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.008037885,0.003546554,0.1156312,0.001933839,0.002621951,0.001232853,0.002995183,0.5490549,0.030523,0.01199757,0.0203874,0.2520376],"study_design_scores_gemma":[0.000172946,0.0009172425,0.008643008,0.00004680366,0.0001638887,0.0002179247,0.0003434962,0.9684803,0.0138196,0.005607472,0.001496343,0.0000910262],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8473881,0.001491608,0.1259152,0.001536027,0.0004361013,0.001096565,0.003497639,0.01325231,0.005386269],"genre_scores_gemma":[0.9184914,0.0001417551,0.07607382,0.0004354705,0.00005892814,0.0002680238,0.003383365,0.0003183599,0.0008288327],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0262284,"threshold_uncertainty_score":0.1387107,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1260834055039945,"score_gpt":0.3864990687183631,"score_spread":0.2604156632143686,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}