torch.manual_seed(123) ​ # reduce d_out_v from 4 to 1, because we have 4 heads d_in, d_out_kq, d_out_v = 3, 2, 4 ​ sa = SelfAttention(d_in, d_out_kq, d_out_v) print(sa(embedded_sentence))