def synchronize_all(): torch.cuda.synchronize() event_queue.join() output_queue.join() with torch.inference_mode(): for i in range(TOTAL_STEPS): if i == WARMUP_STEPS: synchronize_all() start_time = time.perf_counter() profiler.start() elif i == WARMUP_STEPS + PROFILE_STEPS: synchronize_all() profiler.stop() end_time = time.perf_counter() with nvtx.annotate(f"Batch {i}", color="blue"): with nvtx.annotate("get batch", color="red"): batch = next(data_iter) with nvtx.annotate("compute", color="green"): output = model(batch) with nvtx.annotate("copy to CPU", color="yellow"): output_cpu, buf_id = to_cpu(output['out']) with nvtx.annotate("queue CUDA event", color="cyan"): event_pool[buf_id].record() event_queue.put((i, buf_id)) total_time = end_time - start_time throughput = PROFILE_STEPS / total_time print(f"Throughput: {throughput:.2f} steps/sec") # cleanup event_queue.put(None) for _ in range(POSTPROC_WORKERS): output_queue.put(None)