{"id":"W3176904855","doi":"10.1609/aaai.v35i15.17608","title":"On the Softmax Bottleneck of Recurrent Language Models","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Western Canada Research Grid; Compute Canada","keywords":"Perplexity; Softmax function; Bottleneck; Rank (graph theory); Computer science; Monotonic function; Word (group theory); Correlation; Language model; Mathematics; Artificial intelligence; Artificial neural network; Combinatorics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005147085,0.0001786896,0.0002360537,0.00006355246,0.000129173,0.0001371419,0.002022797,0.0000677048,0.00007445806],"category_scores_gemma":[0.0006189284,0.0001131119,0.0001402356,0.0004974843,0.000156016,0.0002336404,0.0005009451,0.0003123216,0.00003369479],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003274399,"about_ca_system_score_gemma":0.0001443957,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002798689,"about_ca_topic_score_gemma":0.000009656372,"domain_scores_codex":[0.9981619,0.00002978717,0.0004983538,0.0004493956,0.0005907054,0.0002698575],"domain_scores_gemma":[0.9982218,0.0002116774,0.0003249493,0.0005599526,0.0006265361,0.00005514087],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00001654145,0.0001026005,0.000009180014,0.00002548739,0.00000992739,6.00273e-7,0.002286291,0.0005605461,0.02154776,0.924757,0.0001045629,0.05057953],"study_design_scores_gemma":[0.00001183553,0.00006511924,0.000009637326,0.0002039011,0.000005225384,0.000002155483,0.0006382797,0.2159328,0.466358,0.3166667,0.00001507753,0.00009126271],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7041706,0.0002277509,0.2379847,0.01607922,0.001115211,0.0006260085,0.00001287227,0.0001300391,0.03965355],"genre_scores_gemma":[0.9960448,0.0000406641,0.003262099,0.0003385804,0.00004344714,0.00001571702,2.47284e-7,0.000008862143,0.0002456252],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6080903,"threshold_uncertainty_score":0.461257,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1059550927922049,"score_gpt":0.2987821668527856,"score_spread":0.1928270740605807,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}