Inside NVIDIA’s cuDNN Graph API: Fusion, Autotuning, and Plan Reuse with cuDNN Frontend
In this tutorial, we work through the cuDNN Frontend‘s graph API from below the framework: we describe a computation as a graph of operations, let cuDNN pick an engine to run it, and then take control of that choice ourselves. Every kernel we build here is expressed the same way: we declare tensors by their dimensions and strides, chain operations onto them, run the five-step build pipeline of validate, build operation graph, create execution plans, check support, and build plans, and then execute against a variant pack of pointers. We run it all on a single Colab GPU, checking each result against a PyTorch reference so we can see both that the fusion is correct and what it costs. The topics build on each other, moving from a single fused convolution to autotuning across engine configs, FP8-style epilogues, attention, plan serialization, dynamic shapes, and CUDA graph capture. Copy CodeCopiedUse a different Browser import os import sys import glob import math import time import ctypes import traceback import subprocess RESULTS = {} def banner(title): print(“n” + “=” * 78) print(title) print(“=” * 78) def section(name): def wrap(fn): def run(*a, **kw): banner(name) try: out = fn(*a, **kw) RESULTS[name] = out if isinstance(out, str) else “ok” return out except Exception as e: RESULTS[name] = f”SKIPPED / FAILED -> {type(e).__name__}: {e}” print(f”n[!] {name} did not complete: {type(e).__name__}: {e}”) traceback.print_exc(limit=3) return None return run return wrap banner(“0. Install nvidia-cudnn-frontend and locate libcudnn”) subprocess.run( [sys.executable, “-m”, “pip”, “install”, “-q”, “nvidia-cudnn-frontend”], check=True, ) import torch assert torch.cuda.is_available(), “No GPU. Runtime -> Change runtime type -> GPU.” torch.backends.cudnn.enabled = True _ = torch.nn.functional.conv2d( torch.randn(1, 1, 8, 8, device=”cuda”), torch.randn(1, 1, 3, 3, device=”cuda”) ) torch.cuda.synchronize() try: import nvidia.cudnn _libdir = os.path.join(os.path.dirname(nvidia.cudnn.__file__), “lib”) os.environ[“CUDNN_PATH”] = os.path.dirname(nvidia.cudnn.__file__) os.environ[“LD_LIBRARY_PATH”] = _libdir + “:” + os.environ.get(“LD_LIBRARY_PATH”, “”) for _so in sorted(glob.glob(os.path.join(_libdir, “libcudnn*.so*”))): try: ctypes.CDLL(_so, mode=ctypes.RTLD_GLOBAL) except OSError: pass except Exception as _e: print(f” (no pip cuDNN package found, relying on system cuDNN: {_e})”) import cudnn print(” cuDNN frontend imported successfully.”) banner(“1. Environment”) DEV = torch.device(“cuda”) MAJOR, MINOR = torch.cuda.get_device_capability() SM = MAJOR * 10 + MINOR CUDNN_VER = cudnn.backend_version() print(f” GPU : {torch.cuda.get_device_name(0)}”) print(f” Compute capability : sm_{SM}”) print(f” Torch / CUDA : {torch.__version__} / {torch.version.cuda}”) print(f” cuDNN backend : {CUDNN_VER}”) try: print(f” cuDNN version str : {cudnn.backend_version_string()}”) except Exception: pass DTYPE = torch.bfloat16 if SM >= 80 else torch.float16 HAS_SDPA = SM >= 80 print(f” Working dtype : {DTYPE}”) print(f” Fused SDPA usable : {HAS_SDPA}”) HANDLE = cudnn.create_handle() TORCH2CUDNN = { torch.float16: cudnn.data_type.HALF, torch.bfloat16: cudnn.data_type.BFLOAT16, torch.float32: cudnn.data_type.FLOAT, torch.int32: cudnn.data_type.INT32, torch.int64: cudnn.data_type.INT64, torch.int8: cudnn.data_type.INT8, torch.uint8: cudnn.data_type.UINT8, } def tensor_of(graph, t, name): return graph.tensor( name=name, dim=list(t.size()), stride=list(t.stride()), data_type=TORCH2CUDNN[t.dtype], ) def scalar_of(graph, name): return graph.tensor( name=name, dim=[1, 1, 1], stride=[1, 1, 1], data_type=cudnn.data_type.FLOAT, is_pass_by_value=True, ) def build(graph, heur=None, policy=None): heur = heur or [cudnn.heur_mode.A, cudnn.heur_mode.FALLBACK] graph.validate() graph.build_operation_graph() graph.create_execution_plans(heur) graph.check_support() if policy is None: graph.build_plans() else: graph.build_plans(policy) return graph def workspace_for(graph): n = graph.get_workspace_size() return torch.empty(max(n, 1), device=DEV, dtype=torch.uint8) def bench(fn, warmup=10, iters=50): for _ in range(warmup): fn() torch.cuda.synchronize() s, e = torch.cuda.Event(True), torch.cuda.Event(True) s.record() for _ in range(iters): fn() e.record() torch.cuda.synchronize() return s.elapsed_time(e) / iters def tflops(flops, ms): return flops / (ms * 1e-3) / 1e12 def report(tag, ms, flops=None): extra = f” ({tflops(flops, ms):7.2f} TFLOP/s)” if flops else “” print(f” {tag:<34s} {ms:8.3f} ms{extra}”) We start by installing nvidia-cudnn-frontend and solving the problem that trips up most first runs: making libcudnn.so visible to the frontend’s dynamic loader. We force PyTorch to load its bundled cuDNN first and then preload the shared objects explicitly, so the frontend’s own dlopen resolves against a library already resident in the process. We then report the compute capability, pick bfloat16 or float16 accordingly, create the cuDNN handle, and define the helpers for tensor description, graph building, workspace allocation, and event-based benchmarking that the rest of the notebook reuses. Copy CodeCopiedUse a different Browser N, C, H, W = 32, 128, 56, 56 K, R, S = 256, 3, 3 PAD, STR, DIL = 1, 1, 1 P = (H + 2 * PAD – DIL * (R – 1) – 1) // STR + 1 Q = (W + 2 * PAD – DIL * (S – 1) – 1) // STR + 1 CONV_FLOPS = 2 * N * K * P * Q * C * R * S CONV_STATE = {} @section(“2. Fused Conv -> Bias -> ReLU”) def conv_fusion(): x = torch.randn(N, C, H, W, device=DEV, dtype=DTYPE).to(memory_format=torch.channels_last) w = torch.randn(K, C, R, S, device=DEV, dtype=DTYPE).to(memory_format=torch.channels_last) b = torch.randn(1, K, 1, 1, device=DEV, dtype=DTYPE) y = torch.empty(N, K, P, Q, device=DEV, dtype=DTYPE).to(memory_format=torch.channels_last) g = cudnn.pygraph( handle=HANDLE, name=”conv_bias_relu”, io_data_type=TORCH2CUDNN[DTYPE], intermediate_data_type=cudnn.data_type.FLOAT, compute_data_type=cudnn.data_type.FLOAT, ) X = tensor_of(g, x, “X”) Wt = tensor_of(g, w, “W”) Bt = tensor_of(g, b, “bias”) conv = g.conv_fprop( image=X, weight=Wt, padding=[PAD, PAD], stride=[STR, STR], dilation=[DIL, DIL], compute_data_type=cudnn.data_type.FLOAT, ) biased = g.bias(input=conv, bias=Bt) Y = g.relu(input=biased) Y.set_output(True).set_data_type(TORCH2CUDNN[DTYPE]) Y.set_dim(list(y.size())).set_stride(list(y.stride())) t0 = time.perf_counter() build(g) build_ms = (time.perf_counter() – t0) * 1e3 ws = workspace_for(g) pack = {X: x, Wt: w, Bt: b, Y: y} g.execute(pack, ws) torch.cuda.synchronize() ref = torch.relu(torch.nn.functional.conv2d(x, w, bias=b.flatten(), padding=PAD)) err = (y.float() – ref.float()).abs().max().item() scale = ref.float().abs().max().item() print(f” problem : N{N} C{C} {H}x{W} -> K{K} {R}x{S} ({DTYPE})”) print(f” build : {build_ms:.1f} ms workspace: {ws.numel()/1024:.1f} KiB”) print(f” max |err|: {err:.4f} (ref max {scale:.2f}, rel {err/max(scale,1e-9):.2e})”) assert err / max(scale, 1e-9) < 5e-2, “numerical mismatch vs PyTorch” ms_cudnn = bench(lambda: g.execute(pack, ws)) ms_torch = bench(lambda: torch.relu( torch.nn.functional.conv2d(x, w, bias=b.flatten(), padding=PAD))) print() report(“cuDNN FE (single fused kernel)”, ms_cudnn, CONV_FLOPS) report(“PyTorch (conv+bias, then relu)”, ms_torch, CONV_FLOPS) print(f” speedup: {ms_torch/ms_cudnn:.2f}x”) CONV_STATE.update(graph=g, pack=pack, ws=ws, x=x, w=w, b=b, y=y) return f”{ms_cudnn:.3f} ms, {tflops(CONV_FLOPS, ms_cudnn):.1f} TFLOP/s” conv_fusion() We build our first graph, a convolution followed by a bias add and a ReLU, all fused into a single kernel. We keep every tensor in channels_last because that is what gives cuDNN the NHWC strides its tensor-core engines want, and we pin the output dimensions and strides explicitly so the result is written back in the same layout. We validate the output against torch.nn.functional.conv2d, then benchmark the fused graph against PyTorch running the
Inside NVIDIA’s cuDNN Graph API: Fusion, Autotuning, and Plan Reuse with cuDNN Frontend 投稿を読む »
