import torch.nn.functional as F import math d_k = K.shape[-1] attention_scores = (Q @ K.T) / math.sqrt(d_k) # Lower triangular mask to prevent future token access causal_mask = torch.tril(torch.ones(input_seq_length, input_seq_length)) masked_scores = attention_scores.masked_fill(causal_mask == 0, float('-inf')) attention_weights = F.softmax(masked_scores, dim=-1) output = attention_weights @ V