{"id":"W7125643653","doi":"10.1109/bigdatase66491.2025.00018","title":"Woodpecker: A Locally Deployed Large Language Model for Protecting Sensitive Information via RAG and Semantic Recognition","year":2025,"lang":"","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"123 Certification (Canada)","funders":"","keywords":"Language model; Semantics (computer science); Information model; Natural language; Feature (linguistics); Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001909787,0.001136605,0.001478379,0.0005327711,0.0005747697,0.001230024,0.002688744,0.00197351,0.003733641],"category_scores_gemma":[0.005187975,0.0004789624,0.000943777,0.0003632476,0.001682597,0.003181938,0.003884838,0.003261175,0.002085756],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006697254,"about_ca_system_score_gemma":0.00141735,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002351035,"about_ca_topic_score_gemma":0.003277453,"domain_scores_codex":[0.9988796,0.0003516827,0.0000418299,0.0002698876,0.0003114286,0.0001456636],"domain_scores_gemma":[0.9979791,0.0008572285,0.0001083238,0.0007916829,0.0001749157,0.00008875429],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001151524,0.0003966099,0.0007186245,0.0001733678,0.0002595915,0.0003683647,0.0001618161,0.6034727,0.03217641,0.04962847,0.02383604,0.2876565],"study_design_scores_gemma":[0.00001748825,0.00004913596,0.00003753313,0.000004880133,0.00001115603,0.000033678,0.000008327585,0.9773822,0.004971925,0.01640024,0.001068001,0.00001529195],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01250526,0.0002853238,0.9722998,0.0002937533,0.000148368,0.00007451972,0.0001930658,0.01281419,0.0013856],"genre_scores_gemma":[0.6614555,0.0003016944,0.3196385,0.001299985,0.0001770761,0.0002514285,0.001085709,0.001870258,0.01391982],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003733641,"threshold_uncertainty_score":0.01249027,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01084179017092142,"score_gpt":0.263793288223626,"score_spread":0.2529514980527046,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}