egress_stream = torch.cuda.Stream() with torch.inference_mode(): for i in range(TOTAL_STEPS): if i == WARMUP_STEPS: synchronize_all() start_time = time.perf_counter() profiler.start() elif i == WARMUP_STEPS + PROFILE_STEPS: synchronize_all() profiler.stop() end_time = time.perf_counter() with nvtx.annotate(f"Batch {i}", color="blue"): with nvtx.annotate("get batch", color="red"): batch = next(data_iter) with nvtx.annotate("compute", color="green"): output = model(batch) # on separate stream with torch.cuda.stream(egress_stream): # wait for default stream to complete compute egress_stream.wait_stream(torch.cuda.default_stream()) # Mark that egress_stream will use this tensor output['out'].record_stream(egress_stream) with nvtx.annotate("copy to CPU", color="yellow"): output_cpu, buf_id = to_cpu(output['out']) with nvtx.annotate("queue CUDA event", color="cyan"): event_pool[buf_id].record(egress_stream) event_queue.put((i, buf_id))