import time, torch, nvtx from torch.utils.data import Dataset, DataLoader from torch.cuda import profiler from torchvision.models.segmentation import deeplabv3_resnet50 DEVICE = "cuda" WARMUP_STEPS = 10 PROFILE_STEPS = 3 COOLDOWN_STEPS = 1 TOTAL_STEPS = WARMUP_STEPS + PROFILE_STEPS + COOLDOWN_STEPS BATCH_SIZE = 64 TOTAL_SAMPLES = TOTAL_STEPS * BATCH_SIZE IMG_SIZE = 512 N_CLASSES = 21 NUM_WORKERS = 8 ASYNC_DATALOAD = True # A synthetic Dataset with random images class FakeDataset(Dataset): def __len__(self): return TOTAL_SAMPLES def __getitem__(self, index): img = torch.randn((3, IMG_SIZE, IMG_SIZE)) return img # utility class for prefetching data to GPU class DataPrefetcher: def __init__(self, loader): self.loader = iter(loader) self.stream = torch.cuda.Stream() self.next_batch = None self.preload() def preload(self): try: data = next(self.loader) with torch.cuda.stream(self.stream): next_data = data.to(DEVICE, non_blocking=ASYNC_DATALOAD) self.next_batch = next_data except: self.next_batch = None def __iter__(self): return self def __next__(self): torch.cuda.current_stream().wait_stream(self.stream) data = self.next_batch self.preload() return data model = deeplabv3_resnet50(weights_backbone=None).to(DEVICE).eval() data_loader = DataLoader( FakeDataset(), batch_size=BATCH_SIZE, num_workers=NUM_WORKERS, pin_memory=ASYNC_DATALOAD ) data_iter = DataPrefetcher(data_loader) def synchronize_all(): torch.cuda.synchronize() def to_cpu(output): return output.cpu() def process_output(batch_id, logits): # do some post processing on output with open('/dev/null', 'wb') as f: f.write(logits.numpy().tobytes()) with torch.inference_mode(): for i in range(TOTAL_STEPS): if i == WARMUP_STEPS: synchronize_all() start_time = time.perf_counter() profiler.start() elif i == WARMUP_STEPS + PROFILE_STEPS: synchronize_all() profiler.stop() end_time = time.perf_counter() with nvtx.annotate(f"Batch {i}", color="blue"): with nvtx.annotate("get batch", color="red"): batch = next(data_iter) with nvtx.annotate("compute", color="green"): output = model(batch) with nvtx.annotate("copy to CPU", color="yellow"): output_cpu = to_cpu(output['out']) with nvtx.annotate("process output", color="cyan"): process_output(i, output_cpu) total_time = end_time - start_time throughput = PROFILE_STEPS / total_time print(f"Throughput: {throughput:.2f} steps/sec")