From bd5c4402936ee6728f8b003ea113f394bf25ad1d Mon Sep 17 00:00:00 2001 From: Oseltamivir <58582368+Oseltamivir@users.noreply.github.com> Date: Mon, 28 Sep 2026 12:38:07 +0800 Subject: [PATCH] CollectiveX: run nccl-ep's communicator in NCCL graph usage mode 1 The default graph usage mode (2, mixing) wraps every captured collective in an external event wait and record. Graphed HT decode captures one host collective per pair, the routing ncclAllGather, and the h100 EP8 trace showed ~14us idle before it, making the graphed pair period 1.14x eager. Mode 1 (one graph at a time, never concurrent with uncaptured work on the communicator) removes it. Graphed HT decode rows carry the -gum1 generation suffix. --- collectivex/bench/ep_nccl.py | 21 +++++++++++++++++++-- collectivex/docs/methodology.md | 9 ++++++--- collectivex/tests/test_backends.py | 18 ++++++++++++++++++ 3 files changed, 43 insertions(+), 5 deletions(-) diff --git a/collectivex/bench/ep_nccl.py b/collectivex/bench/ep_nccl.py index e1e4114448..43ffab7104 100644 --- a/collectivex/bench/ep_nccl.py +++ b/collectivex/bench/ep_nccl.py @@ -90,6 +90,7 @@ class NCCLEPBackend(EPBackend): # pool with post-upgrade rows. # "-static" marks the combine input bound to the full static receive plane (see # `_bind_ht_recv_count`); "-zc" rows before it sliced that input to the received count. + # Graphed HT decode appends "-gum1": its communicator runs graph usage mode 1 (`_comm_config`). kernel_generation = "nccl-ep-v02-ht-routed-zc-static" SUPPORTED_MODES = ("normal", "low-latency") CUDA_GRAPH_MODES = ("normal", "low-latency") @@ -99,7 +100,7 @@ class NCCLEPBackend(EPBackend): @property def cuda_graph_supported(self) -> bool: # HT replays at decode only, as a captured decode step runs it (engines run prefill - # uncaptured). It is slower there at EP16: see methodology, CUDA Graph Replay. + # uncaptured); see methodology, CUDA Graph Replay. if self.mode == "normal" and getattr(self.args, "phase", None) != "decode": return False return super().cuda_graph_supported @@ -143,6 +144,10 @@ def __init__(self, args, rank, world_size, local_rank, device): # NCCL LL rank-major suppresses duplicate ranks in top-k order, accumulates the # returned BF16 rank rows in FP32, and narrows only at the final output. self.combine_reduction = "rank-fp32" + elif self.cuda_graph_supported: + # Only graphed HT captures a host NCCL collective (the routing ncclAllGather), so + # only its rows change with the communicator's graph usage mode. + self.kernel_generation = f"{type(self).kernel_generation}-gum1" # NCCL EP's handle is explicitly reusable across dispatch/combine cycles (ep_test.py # cached mode redispatches and recombines on one handle), so — unlike DeepEP's legacy # low-latency Buffer — no timed component needs a fresh dispatch or a draining combine; @@ -220,9 +225,21 @@ def _bootstrap_comm(self): n = int(length.item()) uid = nccl_core.UniqueId.from_bytes(bytes(payload[:n].cpu().numpy().tobytes())) self._comm = nccl_core.Communicator.init( - nranks=self.world_size, rank=self.rank, unique_id=uid + nranks=self.world_size, rank=self.rank, unique_id=uid, config=self._comm_config() ) + @staticmethod + def _comm_config(): + """Graph usage mode 1: one graph at a time on this communicator, never concurrent with + uncaptured work on it -- how the harness and a captured decode step both use it. + + The default (2, "mixing") wraps every captured collective in an external event wait and + an event record so graph and uncaptured work can interleave (NCCL strongstream.cc). Traced + on h100 HT decode EP8, that left ~14us idle before each routing ncclAllGather and made the + graphed pair period 1.14x eager; mode 0/1 removes the gap (1.03x at T=1, 1.00x at T=256). + """ + return nccl_core.NCCLConfig(graph_usage_mode=1) + # ---- buffer construction ----------------------------------------------------------------- def create_buffer(self, spec): diff --git a/collectivex/docs/methodology.md b/collectivex/docs/methodology.md index 5ca666e4c3..3433e2ae27 100644 --- a/collectivex/docs/methodology.md +++ b/collectivex/docs/methodology.md @@ -313,9 +313,12 @@ under `CUDAGraph.replay()` by default: each library's best measured configuratio every check, without changing its contract. - **nccl-ep** low-latency and HT **decode**. HT decode replays as a captured decode step would - run it, although it is 3-11% slower than eager at h100/h200 EP16: captured, the per-step routing - `ncclAllGather` is a proxy-driven cross-node collective that NCCL fronts with a host-callback - node on every replay (about +50 µs of dispatch; nothing within one node). HT prefill stays eager. + run it. Its per-step routing `ncclAllGather` is the one host NCCL collective in any capture, so + nccl-ep's communicator runs NCCL graph usage mode 1 (one graph at a time): the default mixing + mode wraps every captured collective in an external event wait and record, which left ~14 µs + idle before each all-gather (h100 EP8 trace). Rows carry the `-gum1` generation suffix. Across + nodes the captured all-gather is still proxy-driven, and NCCL fronts it with a host-callback + node on every replay (about +50 µs of dispatch at EP16). HT prefill stays eager. - **flashinfer-ep** decode; prefill stays eager (graphs change nothing there). - **uccl-ep** low-latency, intranode only, except b200 FP8 (faster eager). Normal mode host-syncs. - **deepep-v2** low-latency and normal **decode**, run as vLLM's graphed `deepep_v2` decode runs diff --git a/collectivex/tests/test_backends.py b/collectivex/tests/test_backends.py index a15121218d..be6592c19c 100644 --- a/collectivex/tests/test_backends.py +++ b/collectivex/tests/test_backends.py @@ -209,6 +209,24 @@ def test_only_ll_rank_major_selects_the_nccl_fp32_reduction(self): self.assertEqual(ll.combine_reduction, "rank-fp32") self.assertEqual(getattr(ht, "combine_reduction", "domain-fp32"), "domain-fp32") + def test_graphed_ht_decode_runs_graph_usage_mode_one(self): + module = self._module() + module.dist.group = types.SimpleNamespace(WORLD=object()) + + def base_init(instance, options, rank, world_size, local_rank, device): + instance.args = options + instance.mode = options.mode + + common = dict(experts=384, hidden=7168, topk=6, scale_up_domain=8, mode="normal") + with mock.patch.object(module.EPBackend, "__init__", base_init), \ + mock.patch.dict(os.environ, {}, clear=True): + decode = module.NCCLEPBackend(types.SimpleNamespace(phase="decode", **common), 0, 16, 0, "cuda:0") + prefill = module.NCCLEPBackend(types.SimpleNamespace(phase="prefill", **common), 0, 16, 0, "cuda:0") + self.assertEqual(decode.kernel_generation, "nccl-ep-v02-ht-routed-zc-static-gum1") + self.assertEqual(prefill.kernel_generation, "nccl-ep-v02-ht-routed-zc-static") + with mock.patch.object(module.nccl_core, "NCCLConfig", lambda **kw: kw, create=True): + self.assertEqual(module.NCCLEPBackend._comm_config(), {"graph_usage_mode": 1}) + def test_ll_layout_selector_restores_the_expert_major_contract(self): module = self._module() em = self._construct(module, 8, COLLX_NCCL_LL_LAYOUT="expert-major")