{"id":"W2626778328","doi":"10.65215/2q58a426","title":"Attention Is All You Need","year":2025,"lang":"en","type":"preprint","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6569,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Machine translation; Transformer; BLEU; Encoder; Artificial intelligence; Parallelizable manifold; Parsing; Language model; Natural language processing; Decoding methods; Task (project management); Convolutional neural network; Speech recognition; Machine learning; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001290036,0.000918586,0.0005766302,0.0008549188,0.002028028,0.004122976,0.001106675,0.002498686,0.1262392],"category_scores_gemma":[0.008535541,0.0004529104,0.0006130619,0.0008899282,0.001891009,0.01017581,0.003955811,0.003710305,0.06712462],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001192543,"about_ca_system_score_gemma":0.001253329,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004577928,"about_ca_topic_score_gemma":0.006797184,"domain_scores_codex":[0.9988675,0.0002697808,0.00004314049,0.0002760145,0.0003324497,0.00021119],"domain_scores_gemma":[0.9979519,0.0006074531,0.0001224088,0.000474212,0.0004725642,0.0003715249],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001585597,0.00006500304,0.001613597,0.0002840811,0.00004232508,0.0003149215,0.001550167,0.0005070161,0.002492619,0.07725223,0.6155868,0.3001327],"study_design_scores_gemma":[0.00001312493,0.00002882086,0.0009472517,0.0001700805,0.00002948092,0.0003914603,0.000664629,0.0009713371,0.0009812773,0.0611787,0.9345853,0.00003852327],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"methods","genre_scores_codex":[0.01293748,0.009356464,0.1032492,0.1765424,0.009979519,0.0001821989,0.0029446,0.01012203,0.6746862],"genre_scores_gemma":[0.2172754,0.00843675,0.04344456,0.07747998,0.005234098,0.0003113099,0.003860049,0.005544854,0.6384131],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.1262392,"threshold_uncertainty_score":0.4223121,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01958296515034699,"score_gpt":0.3049060813157011,"score_spread":0.2853231161653542,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}