# define two CUDA streams compute_stream = torch.cuda.Stream() copy_stream = torch.cuda.Stream() # extract first batch next_batch = next(data_iter) with torch.cuda.stream(copy_stream): next_batch = copy_data(next_batch) for i in range(TOTAL_STEPS): if i == WARMUP_STEPS: torch.cuda.synchronize() start_time = time.perf_counter() profiler.start() elif i == WARMUP_STEPS + PROFILE_STEPS: torch.cuda.synchronize() profiler.stop() end_time = time.perf_counter() with nvtx.annotate(f"Batch {i}", color="blue"): # wait for copy stream to complete copy of batch N compute_stream.wait_stream(copy_stream) batch = next_batch # execute model on batch N+1 compute stream try: with nvtx.annotate("get batch", color="red"): next_batch = next(data_iter) with torch.cuda.stream(copy_stream): with nvtx.annotate("copy batch", color="yellow"): next_batch = copy_data(next_batch) except: # reached end of dataset next_batch = None # execute model on batch N compute stream with torch.cuda.stream(compute_stream): with nvtx.annotate("Compute", color="green"): compute_step(model, batch, optimizer) total_time = end_time - start_time throughput = PROFILE_STEPS / total_time print(f"Throughput: {throughput:.2f} steps/sec")