import numpy as np class PositionalEncoder(): """ An implementation of positional encoding. Attributes: d_model (int): The number of embedding dimensions in the learned embeddings. This is used to determine the length of the positional encoding vectors, which make up the rows of the positional encoding matrix. max_length (int): The maximum sequence length in the transformer. This is used to determine the size of the positional encoding matrix. rounding (int): The number of decimal places to round each of the values to in the output positional encoding matrix. """ def __init__(self, d_model, max_length, rounding): self.d_model = d_model self.max_length = max_length self.rounding = rounding def generate_positional_encoding(self): """ Generate positional information to add to inputs for encoding. The positional information is generated using the number of embedding dimensions (d_model), the maximum length of the sequence (max_length), and the number of decimal places to round to (rounding). The output matrix generated is of size (max_length X embedding_dim), where each row is the positional information to be added to the learned embeddings, and each column is an embedding dimension. """ position = np.arange(0, self.max_length).reshape(self.max_length, 1) even_i = np.arange(0, self.d_model, 2) denominator = 10_000**(even_i / self.d_model) even_encoded = np.round(np.sin(position / denominator), self.rounding) odd_encoded = np.round(np.cos(position / denominator), self.rounding) # Interleave the even and odd encodings positional_encoding = np.stack((even_encoded, odd_encoded),2) .reshape(even_encoded.shape[0],-1) # If self.d_model is odd remove the extra column generated if self.d_model % 2 == 1: positional_encoding = np.delete(positional_encoding, -1, axis=1) return positional_encoding def encode(self, input): """ Encode the input by adding positional information. Args: input (np.array): A two-dimensional array of embeddings. The array should be of size (self.max_length x self.d_model). Returns: output (np.array): A two-dimensional array of embeddings plus the positional information. The array has size (self.max_length x self.d_model). """ positional_encoding = self.generate_positional_encoding() output = input + positional_encoding return output MAX_LENGTH = 5 EMBEDDING_DIM = 3 ROUNDING = 2 # Instantiate the encoder PE = PositionalEncoder(d_model=EMBEDDING_DIM, max_length=MAX_LENGTH, rounding=ROUNDING) # Create an input matrix of word embeddings without positional encoding input = np.round(np.random.rand(MAX_LENGTH, EMBEDDING_DIM), ROUNDING) # Create an output matrix of word embeddings by adding positional encoding output = PE.encode(input) # Print the results print(f'Embeddings without positional encoding:nn{input}n') print(f'Positional encoding:nn{output-input}n') print(f'Embeddings with positional encoding:nn{output}')