# Convert scores to probabilities attention_weights = F.softmax(scores, dim=-1) print(f"Attention weights shape: {attention_weights.shape}") # [1, 8, 6, 6] # Apply attention to values attended_values = torch.matmul(attention_weights, V) # [1, 8, 6, 8] # Concatenate heads and project back attended_values = attended_values.transpose(1, 2).contiguous().view( batch_size, seq_len, d_model) # [1, 6, 64] print(f"Final attended values shape: {attended_values.shape}") # Show attention pattern for "cat" print("\nAttention pattern for 'cat':") cat_weights = attention_weights[0, 0, 1, :] # First head for i, word in enumerate(words): print(f" {word}: {cat_weights[i]:.3f}")