attention_output = multi_head_attention(layer_norm(x)) x = x + attention_output ffn_output = feed_forward(layer_norm(x)) output = x + ffn_output