{"id":"W4415262028","doi":"10.48550/arxiv.2510.12117","title":"Locket: Robust Feature-Locking Technique for Language Models","year":2025,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Government of Ontario","keywords":"Evasion (ethics); Scheme (mathematics); Scalability; Language model; Robustness (evolution); Backdoor; Credential; Chatbot","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004677056,0.00101309,0.000770451,0.0009267199,0.0009415353,0.001991534,0.003693023,0.001963075,0.008684212],"category_scores_gemma":[0.02475887,0.0007538606,0.001688541,0.0006670287,0.001994018,0.008632104,0.005684643,0.004044472,0.003153032],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001615036,"about_ca_system_score_gemma":0.001872022,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002000524,"about_ca_topic_score_gemma":0.002524059,"domain_scores_codex":[0.9963253,0.001341199,0.0002758939,0.0005792539,0.001150215,0.000328166],"domain_scores_gemma":[0.9883868,0.004886513,0.0008269088,0.004917226,0.0006952865,0.0002873064],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001169178,0.0004699034,0.004203537,0.000542133,0.000232858,0.0007121849,0.001489884,0.2053097,0.02716359,0.2907121,0.0301944,0.4378004],"study_design_scores_gemma":[0.00006077105,0.000115924,0.0001417419,0.00003779796,0.00002584555,0.0002152584,0.00007131846,0.8849797,0.01164639,0.09291209,0.00974786,0.00004532958],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007421113,0.00009444162,0.9767092,0.0003530234,0.00004696135,0.0001299291,0.000205063,0.01356891,0.001471265],"genre_scores_gemma":[0.6212856,0.0001989953,0.3665292,0.0007216904,0.0001146064,0.0005345714,0.0007843971,0.002286216,0.00754489],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008684212,"threshold_uncertainty_score":0.0290516,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08470765334721182,"score_gpt":0.2049773955053983,"score_spread":0.1202697421581864,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}