def profile(fn, input, labels): def export_trace(p): p.export_chrome_trace(f"{fn.__name__}.json") with torch.profiler.profile( activities=[torch.profiler.ProfilerActivity.CPU, torch.profiler.ProfilerActivity.CUDA], with_stack=True, schedule=torch.profiler.schedule(wait=0, warmup=10, active=5), on_trace_ready=export_trace ) as prof: for _ in range(20): fn(input, labels) torch.cuda.synchronize() # explicit sync for trace readability prof.step() # create random input input_samples = torch.randn((INPUT_SAMPLES, FEATURE_DIM), device='cuda') labels = torch.randint(0, 2, (INPUT_SAMPLES,), device='cuda', dtype=torch.int64) # run with profiler profile(sample_data, input_samples, labels)