import numpy as np # Step 1: inputs (3 words, each as 2D vector) X = np.array([ [1., 0.], # Word A [0., 1.], # Word B [1., 1.] # Word C ]) # Step 2: queries, keys, values (use identity - no change) Q = X K = X V = X # Step 3: attention scores (dot product Q and K.T) scores = Q @ K.T # Step 4: softmax (row-wise) def softmax(x): e_x = np.exp(x - np.max(x, axis=-1, keepdims=True)) return e_x / e_x.sum(axis=-1, keepdims=True) weights = softmax(scores) # Step 5: attention output (weights @ V) output = weights @ V print("Attention output vectors for A, B, C:") print(output)