# Regular BF16 GEMM (what you normally do) reference = lhs @ rhs.t() # Standard PyTorch GEMM # Check memory usage bf16_memory = lhs.numel() * 2 + rhs.numel() * 2 # 2 bytes per BF16 fp8_memory = lhs.numel() * 1 + rhs.numel() * 1 # 1 byte per FP8 print(f"BF16 memory: {bf16_memory / 1024**2:.1f} MB") print(f"FP8 memory: {fp8_memory / 1024**2:.1f} MB") print(f"Memory saved: {(1 - fp8_memory/bf16_memory)*100:.1f}%")