{"id":"W7106831256","doi":"10.48448/80gk-g470","title":"Improbable Bigrams Expose Vulnerabilities of Incomplete Tokens in Byte-Level Tokenizers","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Bigram; Exploit; Byte; Encoding (memory); Language model; Reduction (mathematics); Set (abstract data type)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001506534,0.0006196311,0.000467541,0.0005391853,0.0004934326,0.001438142,0.001156137,0.0009538161,0.005255111],"category_scores_gemma":[0.01429385,0.0005093795,0.0003914396,0.0004895866,0.001519592,0.004594171,0.002173796,0.001359447,0.001952623],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000503551,"about_ca_system_score_gemma":0.0008687981,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008657472,"about_ca_topic_score_gemma":0.001025711,"domain_scores_codex":[0.9982358,0.0006077007,0.0001838054,0.000302633,0.0005275894,0.0001423646],"domain_scores_gemma":[0.9874134,0.006634878,0.001176895,0.003833178,0.0007332722,0.0002083179],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003660529,0.0004032294,0.02306242,0.001995165,0.0002806805,0.005296078,0.007252878,0.02960358,0.3469036,0.09261052,0.01586017,0.4730711],"study_design_scores_gemma":[0.0001262031,0.0009714606,0.005041171,0.0002928344,0.0002243084,0.004861017,0.001774821,0.2710848,0.5812719,0.07928258,0.05482893,0.0002400514],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5578135,0.001021319,0.4046201,0.001215264,0.0004065339,0.0002526052,0.001416766,0.0223866,0.01086747],"genre_scores_gemma":[0.9310766,0.0002035581,0.06121207,0.0002993055,0.00004208546,0.00009824285,0.00074281,0.002091527,0.004233958],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.005255111,"threshold_uncertainty_score":0.01758009,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0955715015522089,"score_gpt":0.3318146834043647,"score_spread":0.2362431818521558,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}