import numpy as np # Example word features (embeddings) for a sequence of 3 words. word_features = np.array([ [1.0, 0.0], # "Alice" [0.0, 2.0], # "loves" [1.0, 1.0] # "books" ]) # We want the model to focus on the relevant words for "books". query = word_features[2] # "books": [1.0, 1.0] keys = word_features # One key for each word # Score each word by similarity (dot product) to the query. scores = keys @ query # scores: [1, 2, 2] # Normalize scores so they sum to 1 (softmax). exp_scores = np.exp(scores) attention_weights = exp_scores / np.sum(exp_scores) # Weighted sum of word features, using attention weights. context_vector = (attention_weights[:, None] * word_features).sum(axis=0) print("Attention weights:", np.round(attention_weights, 2)) print("Context vector:", np.round(context_vector, 2))