attention_scores = inputs @ inputs.T attention_weights = torch.softmax(attention_scores, dim=-1) context_vectors = attention_weights @ inputs __ __