[ { "file": "fig1-03.png", "page": 3, "label": "Figure 1", "caption": "The Transformer - model architecture." }, { "file": "fig2-04.png", "page": 4, "label": "Figure 2", "caption": "(left) Scaled Dot-Product Attention. (right) Multi-Head Attention consists of several attention layers running in parallel." }, { "file": "table2-08.png", "page": 8, "label": "Table 2", "caption": "The Transformer achieves better BLEU scores than previous state-of-the-art models on the English-to-German and English-to-French newstest2014 tests at a fraction of the training cost." }, { "file": "fig3-13.png", "page": 13, "label": "Figure 3", "caption": "An example of the attention mechanism following long-distance dependencies in the encoder self-attention in layer 5 of 6. Many of the attention heads attend to a distant dependency of the verb 'making', completing the phrase 'making...more difficult'. Attentions here shown only for the word 'making'. Different colors represent different heads. Best viewed in color." } ]