# Attention dimensions d_model = 64 num_heads = 8 d_k = d_model // num_heads # 8 # Linear transformations to create Q, K, V W_q = nn.Linear(d_model, d_model, bias=False) W_k = nn.Linear(d_model, d_model, bias=False) W_v = nn.Linear(d_model, d_model, bias=False) # Transform embeddings Q = W_q(embeddings) # Queries: "What am I looking for?" K = W_k(embeddings) # Keys: "What do I represent?" V = W_v(embeddings) # Values: "What information do I carry?" print(f"Q shape: {Q.shape}") # [6, 64] print(f"K shape: {K.shape}") # [6, 64] print(f"V shape: {V.shape}") # [6, 64]