GPTModel( (tok_emb): Embedding(50257, 768) (pos_emb): Embedding(256, 768) (drop): Dropout(p=0.1) (trf_blocks): Sequential( (0): TransformerBlock( (attn): MultiHeadAttention( (W_query): Linear(768 → 768, bias=False) (W_key): Linear(768 → 768, bias=False) (W_value): Linear(768 → 768, bias=False) (out_proj): Linear(768 → 768, bias=False) (dropout): Dropout(p=0.1) ) (ff): FeedForward( (layers): Sequential( (0): Linear(768 → 3072) (1): GELU() (2): Linear(3072 → 768) ) ) (norm1): LayerNorm() (norm2): LayerNorm() (drop_shortcut): Dropout(p=0.1) ) ... TransformerBlock structure repeated 12 times ... ) (final_norm): LayerNorm() (out_head): Linear(768 → 50257, bias=False) ) __ __