# Attention scores between all word pairs attention_scores = torch.mm(Q, K.transpose(0, 1)) # [10, 10] matrix attention_weights = F.softmax(attention_scores, dim=1) # Each row sums to 1 # Row i, column j = how much word i attends to word j