{"id":"W7102431179","doi":"","title":"Key and Value Weights Are Probably All You Need: On the Necessity of the Query, Key, Value weight Triplet in Encoder-Only and Decoder-Only Transformers","year":2025,"lang":"","type":"article","venue":"ArXiv.org","topic":"Error Correcting Code Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Centre National de la Recherche Scientifique; Institut national de recherche en informatique et en automatique (INRIA); Institute for Catastrophic Loss Reduction","keywords":"Disjoint sets; Weight function; Quadratic equation; Scaling; Function (biology); Transformer; Stability (learning theory)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002264422,0.0008190337,0.000779264,0.0003775051,0.0005554029,0.001981819,0.001714484,0.001292185,0.01088469],"category_scores_gemma":[0.01805879,0.0005610928,0.0006981522,0.0004769226,0.002063653,0.007633362,0.003205875,0.00385298,0.002437104],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001165018,"about_ca_system_score_gemma":0.002025736,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005228152,"about_ca_topic_score_gemma":0.006412984,"domain_scores_codex":[0.9988281,0.0003571796,0.00005334337,0.000307215,0.0002473642,0.0002068321],"domain_scores_gemma":[0.9959591,0.002106732,0.0002124329,0.001166744,0.0003740348,0.0001810726],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001321869,0.0002778262,0.004353653,0.0004279733,0.0001110504,0.0006873242,0.000907255,0.1639745,0.02950636,0.4393256,0.02098903,0.3381176],"study_design_scores_gemma":[0.00006927331,0.0001185585,0.0006787461,0.00005700919,0.0000612116,0.0002296121,0.0001784752,0.6501826,0.01781285,0.3265268,0.004057597,0.00002725473],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1617953,0.0005683744,0.804902,0.005167915,0.0002425041,0.0001170263,0.0007605922,0.003644478,0.02280182],"genre_scores_gemma":[0.8819209,0.0003112326,0.1050023,0.0008652823,0.00007955964,0.0001188527,0.0005835767,0.0008426526,0.01027565],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01088469,"threshold_uncertainty_score":0.03641295,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0219917367474046,"score_gpt":0.2580490801467694,"score_spread":0.2360573433993648,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}