t0 = time.perf_counter() spec_out = target.generate( **inputs, max_new_tokens=200, do_sample=False, assistant_model=draft, streamer=streamer, ) spec_time = time.perf_counter() - t0 print(f"\n⏱ With Speculative Decoding:") print(f" Runtime: {spec_time:.2f}s") print(f" Token speed: {len(spec_out[0]) / spec_time:.1f} tok/s") print(f" Speedup over standard decoding: {base_time / spec_time:.2f}x")