import time, statistics def benchmark(func, num_runs=10): # Warmup func() torch.cuda.synchronize() runtimes = [] for _ in range(num_runs): # reset memory stats before each run torch.cuda.empty_cache() torch.cuda.reset_peak_memory_stats() torch.cuda.synchronize() start = time.perf_counter() _ = func() torch.cuda.synchronize() end = time.perf_counter() runtimes.append(end - start) # Get memory allocator stats from last run mem_stats = torch.cuda.memory_stats() allocated_peak = mem_stats.get('allocated_bytes.all.peak', 0) reserved_peak = mem_stats.get('reserved_bytes.all.peak', 0) f_peak = reserved_peak - allocated_peak f_pct = ( 100 * f_peak / reserved_peak if reserved_peak > 0 else 0 ) print(f"\n{'='*60}") print(f"Runtime Results:") print(f" Mean: {statistics.mean(runtimes):.4f}s") print(f" Std: {statistics.stdev(runtimes):.4f}s") print(f" Min: {min(runtimes):.4f}s") print(f" Max: {max(runtimes):.4f}s") print(f"\nMemory Stats:") print(f" Allocated bytes (peak): {allocated_peak / 1e9:.3f} GB") print(f" Reserved bytes (peak): {reserved_peak / 1e9:.3f} GB") print(f" Fragmentation (peak): {f_peak / 1e9:.3f} GB ({f_pct:.1f}%)") print(f"{'='*60}\n") batch_size = 32 for max_seqlen in [100, 200, 400]: print( f"Benchmarking generation with batch size {batch_size} " f"and max sequence length {max_seqlen}..." ) benchmark( lambda: generate_sequence( model, max_seqlen=max_seqlen, batch_size=batch_size ) )