diff --git a/collectivex/bench/ep_deepep_v2.py b/collectivex/bench/ep_deepep_v2.py index 992e43c97f..5b4b3c617a 100644 --- a/collectivex/bench/ep_deepep_v2.py +++ b/collectivex/bench/ep_deepep_v2.py @@ -146,6 +146,9 @@ def __init__(self, args, rank, world_size, local_rank, device): self._normal_cpu_sync = self.mode == "normal" and args.phase != "decode" if self.mode == "normal" and not self._normal_cpu_sync: self.kernel_generation = "v2-elastic-buffer-nosync" + if (os.environ.get("EP_WIN_RELAXED_ORDERING") == "1" and self.mode == "normal" + and world_size > int(args.scale_up_domain)): + self.kernel_generation += "-relaxed-ordering" # its own series; see methodology.md if self.mode == "low-latency": self._enable_ll("legacy-buffer-ll") # the legacy Buffer IBGDA decode kernels diff --git a/collectivex/configs/platform_config.json b/collectivex/configs/platform_config.json index c1e288c384..963e116561 100644 --- a/collectivex/configs/platform_config.json +++ b/collectivex/configs/platform_config.json @@ -42,7 +42,8 @@ "squash_dir": "/home/sa-shared/containers" }, "network": { - "rdma_devices": "mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_6,mlx5_7" + "rdma_devices": "mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_6,mlx5_7", + "rdma_relaxed_ordering": 1 } }, "b200-nscale": { diff --git a/collectivex/docs/methodology.md b/collectivex/docs/methodology.md index 3433e2ae27..4869151384 100644 --- a/collectivex/docs/methodology.md +++ b/collectivex/docs/methodology.md @@ -121,14 +121,16 @@ case count. | GB200/GB300 | 2x4 MNNVL, scale-up | 4x4 MNNVL, scale-up | **A virtualized pool can make a scale-out row measure the hypervisor rather than the fabric.** -h200-dgxc EP16 pays roughly three times the cross-node cost of b300 or h100 on identical topology -and identical traffic, while its EP8 rows are correct. The deficit is confined to the hop. It -sustains ~34 GB/s per node against a nominal 8x400G (~4.2 GB/s per GPU-NIC pair) where bare-metal -h100 reaches wire rate. Reordering the NIC-PE mapping to pair each rank with its socket-local NIC -changed nothing (478µs against a 480µs baseline), which rules the selector out and points at the -GDR path being degraded wholesale inside the guest. The retired b200-dgxc pool showed the same -shape. Treat EP16 rows from a virtualized pool as a lower bound on the hardware until the host's -ACS/IOMMU configuration is confirmed. +h200-dgxc EP16 on deepep-v2 paid three to eight times h100's cross-node cost while nccl-ep on the +same nodes ran at h100 rates. ElasticBuffer registers its GIN window `NCCL_WIN_STRICT_ORDERING`, +which strips PCIe relaxed ordering, and in h200's KVM guests strict-ordered GPU-NIC writes cap the +hop near 10 GB/s per GPU. The DeepEP build patches the registration behind a switch that a pool +opts into with `rdma_relaxed_ordering` in its `network` block. On h200-dgxc, bf16 decode at 512 +tokens per rank drops from 3088µs to 620µs per round trip and the oracle passes in bf16 and fp8. +Upstream made the window strict on purpose (DeepEP #661/#674): with relaxed ordering a GIN signal +can in principle overtake the data it announces. That race has not been reproduced, and a passing +oracle does not rule it out. Those rows carry a `-relaxed-ordering` `kernel_generation` suffix and +never pool with the strict series; every other pool keeps upstream's registration. Physical host count does not define scope. Both GB cells remain inside one 72-GPU MNNVL scale-up domain. diff --git a/collectivex/runtime/common.sh b/collectivex/runtime/common.sh index 6a7f59e3bf..a4c1c262b6 100644 --- a/collectivex/runtime/common.sh +++ b/collectivex/runtime/common.sh @@ -27,7 +27,7 @@ COLLX_DEEPEP_V2_TORCH_SPEC="torch==2.11.0" # Build-recipe generation for the DeepEP venv cache key: bump when build flags change without # any pin changing, so venvs that already carry .ready are not reused with a stale recipe. -COLLX_DEEPEP_V2_BUILD_GEN="dlarch1" +COLLX_DEEPEP_V2_BUILD_GEN="dlarch2" COLLX_UCCL_REPO="https://github.com/uccl-project/uccl" COLLX_UCCL_COMMIT="fc1b582031221645ea9fce58aeb57187713145e3" @@ -128,6 +128,7 @@ collx_load_operator_config() { unset COLLX_EXCLUDE_NODES COLLX_NODELIST COLLX_LOCK_DIR COLLX_MASTER_PORT unset COLLX_SOCKET_IFNAME COLLX_RDMA_DEVICES COLLX_IB_GID_INDEX COLLX_RDMA_SERVICE_LEVEL unset COLLX_RDMA_TRAFFIC_CLASS COLLX_RAIL_ISOLATED COLLX_SINGLE_NODE_RDMA_DEVICES COLLX_RDMA_FABRIC + unset COLLX_RDMA_RELAXED_ORDERING unset MASTER_ADDR MASTER_PORT RANK WORLD_SIZE LOCAL_RANK LOCAL_WORLD_SIZE config_path="${COLLECTIVEX_OPERATOR_CONFIG:-${XDG_CONFIG_HOME:-${HOME}/.config}/inferencex/collectivex.json}" if [ ! -e "$config_path" ]; then @@ -307,6 +308,9 @@ collx_apply_network_profile() { # black-hole at QP RTR. [ "$COLLX_RAIL_ISOLATED" != 1 ] || export NCCL_CROSS_NIC=0 fi + # Opts the patched DeepEP GIN window out of strict ordering (deepep_install). + unset EP_WIN_RELAXED_ORDERING + [ "${COLLX_RDMA_RELAXED_ORDERING:-0}" != 1 ] || export EP_WIN_RELAXED_ORDERING=1 if [ -n "${COLLX_IB_GID_INDEX:-}" ]; then [[ "$COLLX_IB_GID_INDEX" =~ ^[0-9]+$ ]] && [ "$COLLX_IB_GID_INDEX" -le 255 ] \ || collx_die "invalid private IB GID index" @@ -1056,8 +1060,8 @@ collx_run_shard() { collx_log "case[$((ci + 1))/$expected_cases] $COLLX_BENCH ranks=$NGPUS" runtime_log="$(collx_private_log_path "runtime-c$(printf '%03d' "$ci")")" # A hang guard, not a work budget: FP8 prefill and multi-node EP16 prefill on pools with - # degraded GPU-NIC p2p (~34 GB/s per node; see docs/methodology.md) legitimately run past 30 - # minutes. 5400 stays inside the 300-minute allocation. + # slow GPU-NIC p2p (see docs/methodology.md) legitimately run past 30 minutes. 5400 stays + # inside the 300-minute allocation. if ! timeout -k 30 "${COLLX_RUN_TIMEOUT:-5400}" \ srun --jobid="$JOB_ID" --nodes="$NODES" \ --ntasks="$NGPUS" --ntasks-per-node="$GPN" --chdir=/tmp \ diff --git a/collectivex/runtime/config.py b/collectivex/runtime/config.py index e3b0157586..1a519c7ff5 100644 --- a/collectivex/runtime/config.py +++ b/collectivex/runtime/config.py @@ -16,6 +16,7 @@ NETWORK_FIELDS = { "socket_ifname", "rdma_devices", "ib_gid_index", "rdma_service_level", "rdma_traffic_class", "rail_isolated", "single_node_rdma_devices", "rdma_fabric", + "rdma_relaxed_ordering", } # Timing knobs, in the order the legacy colon-string encoded them, paired with the run_ep flag # each one drives. The names match configs/sweep.json so a case's timing block is readable diff --git a/collectivex/runtime/prepare_backend.sh b/collectivex/runtime/prepare_backend.sh index 95427b9c20..6623c7fdbf 100644 --- a/collectivex/runtime/prepare_backend.sh +++ b/collectivex/runtime/prepare_backend.sh @@ -270,6 +270,14 @@ deepep_install() { || { collx_log "ERROR: DeepEP V2 environment activation failed"; return 1; } collx_materialize_source "deepep-v2-$COLLX_DEEPEP_V2_COMMIT" "$source_dir" \ || { collx_log "ERROR: DeepEP V2 staged source is invalid"; return 1; } + # Upstream registers the GIN window NCCL_WIN_STRICT_ORDERING, which drops PCIe relaxed ordering + # whatever NCCL_IB_PCI_RELAXED_ORDERING says; EP_WIN_RELAXED_ORDERING=1 opts a pool out. Unset, + # the registration is upstream's. A pin that moves or duplicates the flag fails here. + local window_cu="$source_dir/csrc/kernels/backend/nccl.cu" + local relaxed='get_env("EP_WIN_RELAXED_ORDERING", 0) ? NCCL_WIN_DEFAULT :' + [ "$(grep -c NCCL_WIN_STRICT_ORDERING "$window_cu")" = 1 ] \ + && sed -i "s/NCCL_WIN_STRICT_ORDERING/($relaxed &)/" "$window_cu" \ + || { collx_log "ERROR: DeepEP V2 window-ordering patch failed"; return 1; } # The RDC device-link step (nvcc -dlink) gets no -gencode from the extension build, so nvcc # falls back to its default arch (sm_75 on CUDA 13) and links kernels that cannot load on the # target GPU (gb300/sm103: cudaErrorUnknown). NVCC_PREPEND_FLAGS reaches the dlink too. diff --git a/collectivex/tests/test_backends.py b/collectivex/tests/test_backends.py index be6592c19c..a0a9f6d7eb 100644 --- a/collectivex/tests/test_backends.py +++ b/collectivex/tests/test_backends.py @@ -608,6 +608,20 @@ def test_flashinfer_and_nccl_ht_graph_decode_only(self): self.assertTrue(_gate(nccl, "low-latency", phase="decode").cuda_graph_supported) +class GinWindowOrdering(unittest.TestCase): + """Relaxed-ordering rows must not merge into the strict-ordered series they would replace.""" + + def test_only_relaxed_scale_out_rows_carry_the_suffix(self): + module = _import_stubbed("ep_deepep_v2", deep_ep=_deep_ep("ElasticBuffer", "Buffer")) + relaxed = {"EP_WIN_RELAXED_ORDERING": "1"} + for world_size, env, suffixed in ((16, relaxed, True), (16, {}, False), (8, relaxed, False)): + with self.subTest(world_size=world_size, env=env), \ + mock.patch.dict(os.environ, env, clear=True): + backend = module.DeepEPV2Backend( + args(scale_up_domain=8, runner="h200-dgxc"), 0, world_size, 0, "cpu") + self.assertEqual(backend.kernel_generation.endswith("-relaxed-ordering"), suffixed) + + class LowLatencyCapDecoupling(unittest.TestCase): """The LL receive size and the measured ladder must stay two numbers -- sizing the receive from `max(ladder)` would shift every rung. Driven through the adapter with deep_ep stubbed, diff --git a/collectivex/tests/test_runtime.py b/collectivex/tests/test_runtime.py index 5cebe5a248..0dd6c622dc 100644 --- a/collectivex/tests/test_runtime.py +++ b/collectivex/tests/test_runtime.py @@ -245,6 +245,20 @@ def test_nvshmem_hca_list_per_placement(self) -> None: self.assertEqual(stdout.strip().splitlines()[-1], expected) +class RelaxedOrderingProfileTests(unittest.TestCase): + def test_only_the_registry_flag_relaxes_the_deepep_window(self) -> None: + for flag, expected in (("1", "1"), ("0", "unset"), ("", "unset")): + with self.subTest(flag=flag): + stdout = run_common( + "export COLLX_RDMA_DEVICES=mlx5_0:1 EP_WIN_RELAXED_ORDERING=stale" + f" COLLX_RDMA_RELAXED_ORDERING='{flag}';" + " collx_apply_network_profile 2 nvlink-rdma;" + ' echo "${EP_WIN_RELAXED_ORDERING:-unset}"', + check=True, + ).stdout + self.assertEqual(stdout.strip().splitlines()[-1], expected) + + class StageTests(unittest.TestCase): def test_create_copy_and_validate_cleanup(self) -> None: with tempfile.TemporaryDirectory() as directory: