From 78b4dba1b823c74e8f55dbe6b1590e6125d1f55c Mon Sep 17 00:00:00 2001 From: Natalia Gimelshein Date: Tue, 29 Sep 2026 20:48:51 -0700 Subject: [PATCH 1/2] Use inline CUDA graph annotations in profiler tutorial Export annotations directly with export_chrome_trace and document graph lane options. Remove the annotation pickle and custom Perfetto conversion workflow, require PyTorch 2.15, and release captured NCCL work before process-group cleanup. --- .../cuda_graph_annotations_tutorial.py | 422 ++++-------------- 1 file changed, 91 insertions(+), 331 deletions(-) diff --git a/advanced_source/cuda_graph_annotations_tutorial.py b/advanced_source/cuda_graph_annotations_tutorial.py index 61772144191..a9be87c7f1b 100644 --- a/advanced_source/cuda_graph_annotations_tutorial.py +++ b/advanced_source/cuda_graph_annotations_tutorial.py @@ -14,7 +14,7 @@ * How to capture CUDA graphs with kernel annotations * How to profile annotated graphs - * How to post-process traces with semantic kernel lanes + * How to export traces with inline annotations and semantic kernel lanes * How to visualize graph execution with custom stream assignments * How to annotate communication collectives with the metadata (collective type, message size, group, rank) that eager NCCL @@ -23,11 +23,10 @@ .. grid-item-card:: :octicon:`list-unordered;1em;` Prerequisites :class-card: card-prerequisites - * PyTorch 2.12+ + * PyTorch 2.15+ (use a nightly build until 2.15 is released) * CUDA-capable GPU * Driver/CUDA-compat >= 13.1 for annotation support * cuda-bindings >= 13.1.0 - * perfetto (``pip install perfetto``) CUDA graphs are a powerful optimization technique that can significantly reduce kernel launch overhead by capturing and replaying sequences of CUDA operations. @@ -35,8 +34,8 @@ making it difficult to understand the logical structure of your computation. This tutorial demonstrates how to use **kernel annotations** to add semantic -labels to kernels within CUDA graphs. These annotations can be merged back into -profiler traces to create custom visualization lanes, making it easier to +labels to kernels within CUDA graphs. The profiler can include these annotations +directly when exporting traces and create custom visualization lanes, making it easier to understand and debug complex graph executions. Annotations are not limited to compute kernels. One of the most valuable uses @@ -63,7 +62,7 @@ # # 1. **Label kernel groups** with meaningful names during capture # 2. **Assign custom stream IDs** for visual organization -# 3. **Merge labels into profiler traces** for semantic visualization +# 3. **Export labels directly in profiler traces** for semantic visualization # # The result is a profiler trace where kernels are labeled and organized by # their function, making it much easier to identify performance bottlenecks @@ -96,11 +95,10 @@ # # For this tutorial, you'll need: # -# - PyTorch 2.12+ +# - PyTorch 2.15+ (use a nightly build until 2.15 is released) # - A CUDA GPU # - Driver/CUDA-compat >= 13.1 for annotation support -# - The ``cuda-bindings`` package >= 13.1.0 (``pip install cuda-python``) -# - The ``perfetto`` package for writing the trace (``pip install perfetto``) +# - The ``cuda-bindings`` package >= 13.1.0 (``pip install "cuda-bindings>=13.1.0"``) # # The cuda-bindings package provides the Python bindings for CUDA runtime APIs. # Version 13.1.0+ is required for the ``cudaGraphNodeGetToolsId`` API that @@ -112,30 +110,19 @@ # still work, but ``mark_kernels`` will be a no-op and no semantic lanes will # appear in the final trace. -import copy -import hashlib +import gzip import json import math import os -import pickle -import sys -from collections import Counter, defaultdict +from collections import Counter from pathlib import Path import torch import torch.distributed as dist import torch.multiprocessing from torch.profiler import profile, ProfilerActivity -from torch.cuda._graph_annotations import ( - get_kernel_annotations, - get_stream_for_pg, - mark_kernels, - _is_tools_id_unavailable, -) -from torch.cuda._annotate_cuda_graph_trace import ( - annotate_trace, - load_trace, -) +from torch.cuda.graph_annotations import get_kernel_annotations, is_available, mark_kernels +from torch.cuda._graph_annotations import get_stream_for_pg ############################################################################### # Building a Model @@ -200,12 +187,14 @@ def forward(): # # The key API is ``mark_kernels()``, which takes a dictionary with: # -# - ``name``: A string label for this kernel group (becomes the lane name) +# - ``name``: A string label stored in the kernel event's ``args`` (also used +# as the lane name when a custom stream is assigned) # - ``stream`` (optional): A virtual stream ID for visualization # # Any CUDA kernels launched within the context will be tagged with these -# annotations. Later, when we post-process the profiler trace, these tags -# will be used to organize kernels into custom lanes. +# annotations. When we export the profiler trace with ``graph_lanes="all"``, +# these tags organize kernels into custom lanes. The stream IDs control only +# visualization; they do not change where the kernels execute. ############################################################################### # Capturing a CUDA Graph with Annotations @@ -242,11 +231,16 @@ def capture_graph_with_annotations(model_fn): # ------------------- # # After capturing the graph, we replay it a few times to warm up, then profile -# subsequent replays. The profiler will record kernel execution times, which -# we'll later merge with our annotations. +# subsequent replays. Pass the recorded annotations directly to +# ``export_chrome_trace()`` to include them in the exported kernel events. +# We also export a raw trace from the same profile for comparison. +# +# ``cuda_graph_annotations`` selects the Python exporter automatically when the +# mapping is nonempty. We explicitly select it for both exports so they also +# work with an empty mapping. Both files can be opened in https://ui.perfetto.dev/. def profile_graph(graph, output_dir): - """Profile graph replays and save the trace.""" + """Profile graph replays and export raw and annotated traces.""" output_dir = Path(output_dir) output_dir.mkdir(exist_ok=True, parents=True) @@ -261,244 +255,44 @@ def profile_graph(graph, output_dir): graph.replay() torch.cuda.synchronize() - # Export the raw trace - trace_path = output_dir / "trace_raw.json.gz" - prof.export_chrome_trace(str(trace_path)) - print(f"Saved raw trace to {trace_path}") + raw_trace_path = output_dir / "trace_raw.json.gz" + prof.export_chrome_trace(str(raw_trace_path), use_python_export=True) + print(f"Saved raw trace to {raw_trace_path}") + + annotations = get_kernel_annotations() + annotated_path = output_dir / "trace_annotated.json.gz" + prof.export_chrome_trace( + str(annotated_path), + use_python_export=True, + cuda_graph_annotations=annotations, + graph_lanes="all" if annotations else "none", + ) + print(f"Saved annotated trace to {annotated_path}") - return trace_path + return raw_trace_path, annotated_path ############################################################################### -# Saving Annotation Metadata -# --------------------------- +# Choosing the Trace Layout +# ------------------------- # -# We need to save the annotation metadata in a pickle file that the -# post-processing tool can discover. The file should be named -# ``kernel_annotations_rank0_fwd_bwd.pkl`` and placed where the trace tool -# can find it. - -def save_annotations(output_dir): - """Save kernel annotations to a pickle file.""" - output_dir = Path(output_dir) - output_dir.mkdir(exist_ok=True, parents=True) - annotations_path = output_dir / "kernel_annotations_rank0_fwd_bwd.pkl" - - annotations = dict(get_kernel_annotations()) - with open(annotations_path, "wb") as f: - pickle.dump(annotations, f) - - print(f"Saved {len(annotations)} annotations to {annotations_path}") - return annotations_path - -############################################################################### -# Post-Processing: Merging Annotations into Traces -# ------------------------------------------------- -# -# The final step is to merge the annotations back into the trace. This involves: -# -# 1. Loading the raw trace and annotations -# 2. Calling ``annotate_trace()`` to apply the annotations -# 3. Emitting a native Perfetto ``.pftrace`` that preserves overlapping kernels -# on their real stream -# -# The result is a trace where kernels are organized by your semantic labels. -# -# **Why a Perfetto protobuf trace (not Chrome JSON)?** A Chrome JSON trace -- -# the format ``torch.profiler.export_chrome_trace`` produces -- has a -# fundamental limitation: a single track (a ``(pid, tid)`` row) can only show -# **properly nested** slices, never crossing/overlapping ones. -# -# Perfetto's native **protobuf** trace (``.pftrace``) solves this -# via the ``TrackDescriptor`` field ``sibling_merge_key``. We split -# overlapping slices across hidden *backing* tracks (so each protobuf -# begin/end stack stays validly nested), then give those backing tracks the -# **same** ``sibling_merge_key`` so the Perfetto UI merges them back into a -# single logical row. Nothing is relocated to a fake stream and no timestamp is -# clamped -- the overlap is shown faithfully on the kernel's real stream. -# -# This converter is adapted from Driss Guessous's `transformer_nuggets -# `_ -# (``transformer_nuggets/utils/track_event.py``); we inline a compact, -# self-contained version here. It needs the ``perfetto`` package -# (``pip install perfetto``). - -def _stable_uuid(*parts): - """A stable 60-bit track UUID derived from its identifying parts.""" - digest = hashlib.sha1(":".join(str(p) for p in parts).encode()).hexdigest() - return int(digest[:15], 16) - - -def _assign_nesting_lanes(slices): - """Split overlapping slices into backing lanes so each lane is nestable. - - A lane only holds slices that are either disjoint or fully contained, so a - begin/end stack on that lane never has crossing slices. Returns - ``(lane_of_index, lane_count)``. The lane is a *backing* track index, not a - user-visible stream -- lanes sharing a stream are merged back in the UI. - """ - order = sorted( - range(len(slices)), - key=lambda i: (slices[i]["ts"], -slices[i]["end"], slices[i]["index"]), - ) - lane_of = {} - lane_end_stacks = [] - for i in order: - s = slices[i] - assigned = None - for lane, stack in enumerate(lane_end_stacks): - while stack and stack[-1] <= s["ts"]: - stack.pop() - # Valid if the lane is free or this slice nests inside the open one. - if not stack or s["end"] <= stack[-1]: - stack.append(s["end"]) - assigned = lane - break - if assigned is None: - lane_end_stacks.append([s["end"]]) - assigned = len(lane_end_stacks) - 1 - lane_of[i] = assigned - return lane_of, len(lane_end_stacks) - - -def _add_debug_annotation(track_event, name, value): - """Carry a Chrome event arg over as a typed Perfetto debug annotation.""" - ann = track_event.debug_annotations.add() - ann.name = str(name) - # bool must be checked before int (bool is a subclass of int in Python). - if isinstance(value, bool): - ann.bool_value = value - elif isinstance(value, int): - ann.int_value = value - elif isinstance(value, float): - ann.double_value = value - elif value is None: - ann.string_value = "null" - elif isinstance(value, str): - ann.string_value = value - else: - ann.legacy_json_value = json.dumps(value, default=str) - - -def write_perfetto_trace(trace, output_path): - """Convert a Chrome JSON trace dict to a native Perfetto ``.pftrace``. - - Each Chrome ``(pid, tid)`` row becomes a ``TrackDescriptor``; each ``ph='X'`` - slice becomes a ``TYPE_SLICE_BEGIN`` / ``TYPE_SLICE_END`` pair. Overlapping - slices are split across backing lanes that share a ``sibling_merge_key`` so - the UI re-merges them onto their real stream. - """ - from perfetto.trace_builder.proto_builder import TraceProtoBuilder - from perfetto.protos.perfetto.trace.perfetto_trace_pb2 import ( - TrackDescriptor, - TrackEvent, - ) - - events = trace["traceEvents"] - - # Collect the process/thread names emitted as metadata ('M') events. - process_names, thread_names = {}, {} - for e in events: - if e.get("ph") == "M": - if e.get("name") == "process_name": - process_names[e.get("pid")] = e.get("args", {}).get("name", "") - elif e.get("name") == "thread_name": - key = (e.get("pid"), e.get("tid")) - thread_names[key] = e.get("args", {}).get("name", "") - - # Group complete ('X') slices by their (pid, tid) track. - slices_by_track = defaultdict(list) - for i, e in enumerate(events): - if e.get("ph") == "X": - ts = float(e.get("ts", 0) or 0) - dur = float(e.get("dur", 0) or 0) - slices_by_track[(e.get("pid"), e.get("tid"))].append( - {"event": e, "index": i, "ts": ts, "end": ts + dur} - ) - - def ts_us_to_ns(value): - return int(round(value * 1000.0)) - - builder = TraceProtoBuilder() - SEQ = 1 - - # One descriptor per process. - for pid in {pid for (pid, _tid) in slices_by_track}: - pkt = builder.add_packet() - desc = pkt.track_descriptor - desc.uuid = _stable_uuid("process", pid) - desc.name = process_names.get(pid, f"process {pid}") - - # One descriptor per backing lane; emit begin/end markers per slice. - markers = [] - for (pid, tid), slices in slices_by_track.items(): - lane_of, lane_count = _assign_nesting_lanes(slices) - name = thread_names.get((pid, tid), f"stream {tid}") - lane_uuids = [] - for lane in range(lane_count): - uuid = _stable_uuid("track", pid, tid, lane) - lane_uuids.append(uuid) - pkt = builder.add_packet() - desc = pkt.track_descriptor - desc.uuid = uuid - desc.parent_uuid = _stable_uuid("process", pid) - desc.name = name - # Multiple lanes for one stream -> merge them into one UI row. - if lane_count > 1: - desc.sibling_merge_behavior = ( - TrackDescriptor.SIBLING_MERGE_BEHAVIOR_BY_SIBLING_MERGE_KEY - ) - desc.sibling_merge_key = f"{pid}:{tid}:{name}" - for i, s in enumerate(slices): - uuid = lane_uuids[lane_of[i]] - markers.append((ts_us_to_ns(s["ts"]), 1, uuid, "begin", s["event"])) - markers.append((ts_us_to_ns(s["end"]), 0, uuid, "end", s["event"])) - - # Begin markers must be ordered before end markers at the same timestamp. - markers.sort(key=lambda m: (m[0], m[1])) - for ts_ns, _rank, uuid, kind, event in markers: - pkt = builder.add_packet() - pkt.timestamp = ts_ns - pkt.trusted_packet_sequence_id = SEQ - track_event = pkt.track_event - track_event.track_uuid = uuid - if kind == "begin": - track_event.type = TrackEvent.TYPE_SLICE_BEGIN - track_event.name = event.get("name", "slice") - if event.get("cat"): - track_event.categories.append(event["cat"]) - for key, value in (event.get("args") or {}).items(): - _add_debug_annotation(track_event, key, value) - else: - track_event.type = TrackEvent.TYPE_SLICE_END - - Path(output_path).write_bytes(builder.serialize()) - return output_path - - -def post_process_trace(raw_trace_path, annotations_path, output_dir): - """Merge annotations into the trace and emit a Perfetto ``.pftrace``.""" - output_dir = Path(output_dir) - - # Load raw trace and annotations - raw_trace = load_trace(raw_trace_path) - with open(annotations_path, "rb") as f: - annotations = pickle.load(f) - - # Make a copy for post-processing - annotated_trace = copy.deepcopy(raw_trace) - - # Apply annotations - num_annotated = annotate_trace(annotated_trace, annotations) - print(f"Annotated {num_annotated} kernels in the trace") - - # Emit a native Perfetto protobuf trace. Overlapping kernels are split onto - # backing lanes that re-merge in the UI -- no kernel is relocated to a fake - # stream and no timestamp is mutated. - annotated_path = output_dir / "trace_annotated.pftrace" - write_perfetto_trace(annotated_trace, annotated_path) - print(f"Saved annotated trace to {annotated_path}") - - return annotated_path, raw_trace, annotated_trace +# ``graph_lanes="all"`` places graphed events with a ``stream`` annotation on +# that display lane. Other graphed events go to ``default_stream`` (7 by +# default). Moved events retain their actual execution stream in +# ``args["original_stream"]``. This groups attention on lane 62 and MLP on +# lane 61 in our example. +# +# To keep the recorded stream layout, omit ``graph_lanes`` or set it to +# ``"none"``. Annotation metadata is still included in each matching event's +# ``args``, and any annotated stream is stored as ``args["annotated_stream"]``. +# This is useful when inspecting concurrency on the original streams. +# +# An empty annotation mapping is treated as no annotations. Since +# ``graph_lanes="all"`` requires a nonempty mapping, the example uses +# ``"none"`` when annotation support is unavailable. +# +# Keep the captured graph alive until export finishes: destroying or resetting +# it removes its entries from the live annotation registry. There is no need +# to save annotations to a separate file or post-process the exported trace. ############################################################################### # Comparing Before and After @@ -536,7 +330,7 @@ def count_lanes(trace): # ------------------------ # # Now let's run the complete workflow: build a model, capture it with -# annotations, profile it, and post-process the trace. +# annotations, profile it, and export the annotated trace. def main(): """End-to-end CUDA graph annotation and profiling demo.""" @@ -545,7 +339,7 @@ def main(): # Check if annotation support is available # PyTorch will log a warning if cuda-bindings version is too old - supported = not _is_tools_id_unavailable() + supported = is_available() print(f"Annotation support available: {supported}") if not supported: print("NOTE: Annotation API not available.") @@ -553,7 +347,7 @@ def main(): print(" - Driver/CUDA-compat < 13.1") print(" - Outdated cuda-bindings (check PyTorch warnings above)") print("Annotations will not be recorded, but the demo will still run.") - print("Kernels will be reassigned to the default lane, not semantic lanes.\n") + print("The exported trace will keep the recorded stream layout.\n") output_dir = Path("traces") @@ -565,22 +359,14 @@ def main(): print("\n2. Capturing CUDA graph with annotations...") graph, output = capture_graph_with_annotations(model_fn) - # Save annotations - print("\n3. Saving annotation metadata...") - annotations_path = save_annotations(output_dir) + print("\n3. Profiling graph replays and exporting traces...") + raw_trace_path, annotated_path = profile_graph(graph, output_dir) - # Profile the graph - print("\n4. Profiling graph replays...") - raw_trace_path = profile_graph(graph, output_dir) - - # Post-process the trace - print("\n5. Post-processing: merging annotations into trace...") - annotated_path, raw_trace, annotated_trace = post_process_trace( - raw_trace_path, annotations_path, output_dir - ) - - # Compare before and after - print("\n6. Comparing traces...") + print("\n4. Comparing traces...") + with gzip.open(raw_trace_path, "rt") as f: + raw_trace = json.load(f) + with gzip.open(annotated_path, "rt") as f: + annotated_trace = json.load(f) compare_traces(raw_trace, annotated_trace) # Summary @@ -589,7 +375,6 @@ def main(): print("="*60) print(f"Raw trace: {raw_trace_path}") print(f"Annotated trace: {annotated_path}") - print(f"Annotations: {annotations_path}") print("\nOpen the annotated trace in https://ui.perfetto.dev/ to visualize") print("the semantic kernel lanes.") print("="*60) @@ -605,17 +390,11 @@ def main(): # 2. Capturing CUDA graph with annotations... # Captured graph with 13 annotated nodes # -# 3. Saving annotation metadata... -# Saved 13 annotations to traces/kernel_annotations_rank0_fwd_bwd.pkl -# -# 4. Profiling graph replays... +# 3. Profiling graph replays and exporting traces... # Saved raw trace to traces/trace_raw.json.gz +# Saved annotated trace to traces/trace_annotated.json.gz # -# 5. Post-processing: merging annotations into trace... -# Annotated 65 kernels in the trace -# Saved annotated trace to traces/trace_annotated.pftrace -# -# 6. Comparing traces... +# 4. Comparing traces... # # ============================================================ # BEFORE annotation - kernels per lane (tid -> count): @@ -631,8 +410,7 @@ def main(): # SUMMARY # ============================================================ # Raw trace: traces/trace_raw.json.gz -# Annotated trace: traces/trace_annotated.pftrace -# Annotations: traces/kernel_annotations_rank0_fwd_bwd.pkl +# Annotated trace: traces/trace_annotated.json.gz # # Open the annotated trace in https://ui.perfetto.dev/ to visualize # the semantic kernel lanes. @@ -655,7 +433,7 @@ def main(): # # Annotations close this gap. By wrapping the collective in ``mark_kernels`` # with the same fields the profiler auto-attaches in eager mode, we manually -# re-attach that metadata to the graphed kernel. After post-processing, a +# re-attach that metadata to the graphed kernel. After export, a # graphed collective reads just like an eager one. The helper below builds the # metadata dict; using the field names the profiler uses in eager # (``In msg nelems``, ``Group size``, ``Process Group Name``, ...) keeps the @@ -748,7 +526,7 @@ def init_pg(rank, world_size): torch.cuda.set_device(rank) def _comm_worker(rank, world_size): - """Per-rank worker: build, capture, profile, and (on rank 0) post-process.""" + """Per-rank worker: build, capture, profile, and (on rank 0) export.""" init_pg(rank, world_size) output_dir = Path("traces_comm") @@ -762,13 +540,11 @@ def _comm_worker(rank, world_size): graph, _ = capture_graph_with_annotations(model_fn) # Every rank participates in the collective during profiling, but only - # rank 0 saves and post-processes the trace. + # rank 0 exports the traces. if rank == 0: - annotations_path = save_annotations(output_dir) - raw_trace_path = profile_graph(graph, output_dir) - annotated_path, _, annotated_trace = post_process_trace( - raw_trace_path, annotations_path, output_dir - ) + _, annotated_path = profile_graph(graph, output_dir) + with gzip.open(annotated_path, "rt") as f: + annotated_trace = json.load(f) # Print the args of the annotated collective kernel(s) to show that the # eager-style metadata is now attached to the graphed comm. @@ -799,6 +575,8 @@ def _comm_worker(rank, world_size): graph.replay() torch.cuda.synchronize() + # Release captured NCCL work after export and before destroying the group. + graph.reset() dist.destroy_process_group() def comm_annotation_demo(): @@ -820,11 +598,9 @@ def comm_annotation_demo(): # # Building compute + collective block... # Capturing CUDA graph with annotations... -# Captured graph with 2 annotated nodes -# Saved 2 annotations to traces_comm/kernel_annotations_rank0_fwd_bwd.pkl +# Captured graph with 3 annotated nodes # Saved raw trace to traces_comm/trace_raw.json.gz -# Annotated 5 kernels in the trace -# Saved annotated trace to traces_comm/trace_annotated.pftrace +# Saved annotated trace to traces_comm/trace_annotated.json.gz # # The all_reduce runs a real NCCL kernel # (``ncclDevKernel_AllReduce_Sum_f32_RING_LL``) across the two ranks: @@ -835,8 +611,8 @@ def comm_annotation_demo(): # Out msg nelems: 1048576 # Group size: 2 # dtype: float32 -# Process Group Name: default -# Process Group Description: default +# Process Group Name: 0 +# Process Group Description: default_pg # Process Group Ranks: [0, 1] # stream: 60 # @@ -845,25 +621,6 @@ def comm_annotation_demo(): # and ranks -- the same fields you would see in an eager trace, now recovered # for a CUDA-graphed collective. This metadata is LOST without annotations. -############################################################################### -# How Overlapping Kernels Are Handled -# ------------------------------------ -# -# Graphed CUDA kernels often overlap slightly, and a single trace track can -# only render properly nested slices. The Perfetto converter handles this -# faithfully: -# -# 1. ``_assign_nesting_lanes()``: For each stream, overlapping slices are split -# across hidden *backing* lanes so that each lane's begin/end stack is validly -# nested. A lane is a backing track index, **not** a user-visible stream. -# -# 2. ``sibling_merge_key``: All backing lanes for one stream are given the same -# merge key, so the Perfetto UI merges them back into a single logical row. -# -# The result: overlaps render correctly on the kernel's **real** stream. No -# kernel is relocated to a fabricated stream, and no timestamp is mutated -- -# unlike the legacy Chrome-JSON workaround, which had to do both. - ############################################################################### # Performance Considerations # --------------------------- @@ -872,7 +629,7 @@ def comm_annotation_demo(): # # - Annotation marking happens during graph capture (one-time cost) # - Graph replay performance is identical to unannotated graphs -# - Post-processing is offline and doesn't affect runtime +# - Annotations are added during trace export, after profiling finishes # # The main cost is the profiling itself, which you would do anyway when # optimizing performance. Annotations simply make the profiler output more @@ -886,7 +643,9 @@ def comm_annotation_demo(): # # - Check that your driver/CUDA-compat >= 13.1 # - Verify that ``enable_annotations=True`` was passed to ``torch.cuda.graph()`` -# - Ensure ``cuda-python`` is installed +# - Ensure ``cuda-bindings>=13.1.0`` is installed +# - Pass ``cuda_graph_annotations=get_kernel_annotations()`` to +# ``export_chrome_trace()`` while the graph is still alive # # **Annotations not showing up in specific kernels?** # @@ -900,7 +659,7 @@ def comm_annotation_demo(): # # CUDA graph kernel annotations provide a powerful way to add semantic # structure to your profiling traces. By marking logical components of your -# model during graph capture and merging these annotations in post-processing, +# model during graph capture and including these annotations during export, # you can create visualizations that make it much easier to understand and # optimize complex CUDA graph executions. # @@ -911,7 +670,8 @@ def comm_annotation_demo(): # - Annotate communication collectives to recover the NCCL metadata # (collective type, message size, group, rank) that CUDA graphs drop but # eager traces expose -# - Post-process traces with ``annotate_trace()`` +# - Pass annotations directly to ``export_chrome_trace()`` +# - Use ``graph_lanes="all"`` to organize graphed kernels into semantic lanes # - View results in https://ui.perfetto.dev/ for intuitive visualization # # This technique is especially valuable for large models with many components, From 3355fc34d4975835d479b7c470aa5f9cd26292ea Mon Sep 17 00:00:00 2001 From: Natalia Gimelshein Date: Tue, 29 Sep 2026 21:16:37 -0700 Subject: [PATCH 2/2] Add a mark_stream example for logical profile lanes Capture two independent projections on annotated CUDA streams with explicit fork and join dependencies. Reuse inline trace export to show automatic logical lane assignment and explain how graph_lanes controls the displayed layout. --- .../cuda_graph_annotations_tutorial.py | 72 ++++++++++++++++++- 1 file changed, 71 insertions(+), 1 deletion(-) diff --git a/advanced_source/cuda_graph_annotations_tutorial.py b/advanced_source/cuda_graph_annotations_tutorial.py index a9be87c7f1b..0068d17a894 100644 --- a/advanced_source/cuda_graph_annotations_tutorial.py +++ b/advanced_source/cuda_graph_annotations_tutorial.py @@ -16,6 +16,7 @@ * How to profile annotated graphs * How to export traces with inline annotations and semantic kernel lanes * How to visualize graph execution with custom stream assignments + * How to use ``mark_stream`` to recover logical stream lanes * How to annotate communication collectives with the metadata (collective type, message size, group, rank) that eager NCCL traces expose but CUDA graphs drop @@ -122,7 +123,7 @@ import torch.multiprocessing from torch.profiler import profile, ProfilerActivity from torch.cuda.graph_annotations import get_kernel_annotations, is_available, mark_kernels -from torch.cuda._graph_annotations import get_stream_for_pg +from torch.cuda._graph_annotations import get_stream_for_pg, mark_stream ############################################################################### # Building a Model @@ -416,6 +417,74 @@ def main(): # the semantic kernel lanes. # ============================================================ +############################################################################### +# Recording Logical Streams with ``mark_stream`` +# ----------------------------------------------- +# +# When your computation already uses multiple CUDA streams, ``mark_stream`` +# switches to a stream and records a logical lane ID for its kernels. This +# helper currently lives in the private ``torch.cuda._graph_annotations`` +# module. Use ``mark_kernels`` with an explicit ``stream`` field when you only +# want to change the display layout, as in the transformer example above. +# +# The block below runs two independent projections on separate CUDA streams, +# then combines their results. Each side stream waits for the current stream +# inside its ``mark_stream`` scope. These waits make the side streams join the +# capture before launching work. The current stream waits for both branches +# before reading their outputs and ending capture. ``mark_stream`` does not +# insert these dependencies for you. + +def build_multistream_block(): + """Create two projections on separate annotated CUDA streams.""" + x = torch.randn(1024, 1024, device="cuda") + left_weight = torch.randn_like(x) + right_weight = torch.randn_like(x) + left = torch.empty_like(x) + right = torch.empty_like(x) + left_stream = torch.cuda.Stream() + right_stream = torch.cuda.Stream() + + def forward(): + current_stream = torch.cuda.current_stream() + with mark_stream(left_stream, "left_projection"): + left_stream.wait_stream(current_stream) + torch.mm(x, left_weight, out=left) + + with mark_stream(right_stream, "right_projection"): + right_stream.wait_stream(current_stream) + torch.mm(x, right_weight, out=right) + + current_stream.wait_stream(left_stream) + current_stream.wait_stream(right_stream) + with mark_kernels("combine"): + return left + right + + return forward + +############################################################################### +# Reuse the capture and profiling helpers to export this graph. With +# ``graph_lanes="all"``, the projections appear on distinct logical lanes +# named ``left_projection`` and ``right_projection``, even if graph replay +# schedules them on different hardware streams. The ``combine`` kernel goes +# to the default lane, 7. Moved events retain ``args["original_stream"]``. +# +# Lane IDs are assigned automatically and may vary depending on which streams +# have already been marked. Reusing the same stream reuses its lane ID. Passing +# the current stream to ``mark_stream`` adds the label without assigning a new +# lane. With ``graph_lanes="none"``, the original layout is preserved and the +# recorded lane IDs appear in ``args["annotated_stream"]`` instead. + +def stream_annotation_demo(): + """Capture and profile the two projections with logical stream lanes.""" + model_fn = build_multistream_block() + graph, output = capture_graph_with_annotations(model_fn) + _, annotated_path = profile_graph(graph, "traces_streams") + return annotated_path + +############################################################################### +# Run ``stream_annotation_demo()`` and open +# ``traces_streams/trace_annotated.json.gz`` in https://ui.perfetto.dev/. + ############################################################################### # Annotating Communication Collectives # ------------------------------------- @@ -666,6 +735,7 @@ def comm_annotation_demo(): # Key takeaways: # # - Use ``mark_kernels()`` to label regions during graph capture +# - Use ``mark_stream()`` to record logical lanes when switching CUDA streams # - Enable annotations with ``enable_annotations=True`` # - Annotate communication collectives to recover the NCCL metadata # (collective type, message size, group, rank) that CUDA graphs drop but