Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions collectivex/bench/ep_deepep_v2.py
Original file line number Diff line number Diff line change
Expand Up @@ -146,6 +146,9 @@ def __init__(self, args, rank, world_size, local_rank, device):
self._normal_cpu_sync = self.mode == "normal" and args.phase != "decode"
if self.mode == "normal" and not self._normal_cpu_sync:
self.kernel_generation = "v2-elastic-buffer-nosync"
if (os.environ.get("EP_WIN_RELAXED_ORDERING") == "1" and self.mode == "normal"
and world_size > int(args.scale_up_domain)):
self.kernel_generation += "-relaxed-ordering" # its own series; see methodology.md
if self.mode == "low-latency":
self._enable_ll("legacy-buffer-ll") # the legacy Buffer IBGDA decode kernels

Expand Down
3 changes: 2 additions & 1 deletion collectivex/configs/platform_config.json
Original file line number Diff line number Diff line change
Expand Up @@ -42,7 +42,8 @@
"squash_dir": "/home/sa-shared/containers"
},
"network": {
"rdma_devices": "mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_6,mlx5_7"
"rdma_devices": "mlx5_0,mlx5_1,mlx5_2,mlx5_3,mlx5_4,mlx5_5,mlx5_6,mlx5_7",
"rdma_relaxed_ordering": 1
}
},
"b200-nscale": {
Expand Down
18 changes: 10 additions & 8 deletions collectivex/docs/methodology.md
Original file line number Diff line number Diff line change
Expand Up @@ -121,14 +121,16 @@ case count.
| GB200/GB300 | 2x4 MNNVL, scale-up | 4x4 MNNVL, scale-up |

**A virtualized pool can make a scale-out row measure the hypervisor rather than the fabric.**
h200-dgxc EP16 pays roughly three times the cross-node cost of b300 or h100 on identical topology
and identical traffic, while its EP8 rows are correct. The deficit is confined to the hop. It
sustains ~34 GB/s per node against a nominal 8x400G (~4.2 GB/s per GPU-NIC pair) where bare-metal
h100 reaches wire rate. Reordering the NIC-PE mapping to pair each rank with its socket-local NIC
changed nothing (478µs against a 480µs baseline), which rules the selector out and points at the
GDR path being degraded wholesale inside the guest. The retired b200-dgxc pool showed the same
shape. Treat EP16 rows from a virtualized pool as a lower bound on the hardware until the host's
ACS/IOMMU configuration is confirmed.
h200-dgxc EP16 on deepep-v2 paid three to eight times h100's cross-node cost while nccl-ep on the
same nodes ran at h100 rates. ElasticBuffer registers its GIN window `NCCL_WIN_STRICT_ORDERING`,
which strips PCIe relaxed ordering, and in h200's KVM guests strict-ordered GPU-NIC writes cap the
hop near 10 GB/s per GPU. The DeepEP build patches the registration behind a switch that a pool
opts into with `rdma_relaxed_ordering` in its `network` block. On h200-dgxc, bf16 decode at 512
tokens per rank drops from 3088µs to 620µs per round trip and the oracle passes in bf16 and fp8.
Upstream made the window strict on purpose (DeepEP #661/#674): with relaxed ordering a GIN signal
can in principle overtake the data it announces. That race has not been reproduced, and a passing
oracle does not rule it out. Those rows carry a `-relaxed-ordering` `kernel_generation` suffix and
never pool with the strict series; every other pool keeps upstream's registration.

Physical host count does not define scope. Both GB cells remain inside one 72-GPU MNNVL scale-up
domain.
Expand Down
10 changes: 7 additions & 3 deletions collectivex/runtime/common.sh
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,7 @@ COLLX_DEEPEP_V2_TORCH_SPEC="torch==2.11.0"

# Build-recipe generation for the DeepEP venv cache key: bump when build flags change without
# any pin changing, so venvs that already carry .ready are not reused with a stale recipe.
COLLX_DEEPEP_V2_BUILD_GEN="dlarch1"
COLLX_DEEPEP_V2_BUILD_GEN="dlarch2"

COLLX_UCCL_REPO="https://github.com/uccl-project/uccl"
COLLX_UCCL_COMMIT="fc1b582031221645ea9fce58aeb57187713145e3"
Expand Down Expand Up @@ -128,6 +128,7 @@ collx_load_operator_config() {
unset COLLX_EXCLUDE_NODES COLLX_NODELIST COLLX_LOCK_DIR COLLX_MASTER_PORT
unset COLLX_SOCKET_IFNAME COLLX_RDMA_DEVICES COLLX_IB_GID_INDEX COLLX_RDMA_SERVICE_LEVEL
unset COLLX_RDMA_TRAFFIC_CLASS COLLX_RAIL_ISOLATED COLLX_SINGLE_NODE_RDMA_DEVICES COLLX_RDMA_FABRIC
unset COLLX_RDMA_RELAXED_ORDERING
unset MASTER_ADDR MASTER_PORT RANK WORLD_SIZE LOCAL_RANK LOCAL_WORLD_SIZE
config_path="${COLLECTIVEX_OPERATOR_CONFIG:-${XDG_CONFIG_HOME:-${HOME}/.config}/inferencex/collectivex.json}"
if [ ! -e "$config_path" ]; then
Expand Down Expand Up @@ -307,6 +308,9 @@ collx_apply_network_profile() {
# black-hole at QP RTR.
[ "$COLLX_RAIL_ISOLATED" != 1 ] || export NCCL_CROSS_NIC=0
fi
# Opts the patched DeepEP GIN window out of strict ordering (deepep_install).
unset EP_WIN_RELAXED_ORDERING
[ "${COLLX_RDMA_RELAXED_ORDERING:-0}" != 1 ] || export EP_WIN_RELAXED_ORDERING=1
if [ -n "${COLLX_IB_GID_INDEX:-}" ]; then
[[ "$COLLX_IB_GID_INDEX" =~ ^[0-9]+$ ]] && [ "$COLLX_IB_GID_INDEX" -le 255 ] \
|| collx_die "invalid private IB GID index"
Expand Down Expand Up @@ -1056,8 +1060,8 @@ collx_run_shard() {
collx_log "case[$((ci + 1))/$expected_cases] $COLLX_BENCH ranks=$NGPUS"
runtime_log="$(collx_private_log_path "runtime-c$(printf '%03d' "$ci")")"
# A hang guard, not a work budget: FP8 prefill and multi-node EP16 prefill on pools with
# degraded GPU-NIC p2p (~34 GB/s per node; see docs/methodology.md) legitimately run past 30
# minutes. 5400 stays inside the 300-minute allocation.
# slow GPU-NIC p2p (see docs/methodology.md) legitimately run past 30 minutes. 5400 stays
# inside the 300-minute allocation.
if ! timeout -k 30 "${COLLX_RUN_TIMEOUT:-5400}" \
srun --jobid="$JOB_ID" --nodes="$NODES" \
--ntasks="$NGPUS" --ntasks-per-node="$GPN" --chdir=/tmp \
Expand Down
1 change: 1 addition & 0 deletions collectivex/runtime/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@
NETWORK_FIELDS = {
"socket_ifname", "rdma_devices", "ib_gid_index", "rdma_service_level",
"rdma_traffic_class", "rail_isolated", "single_node_rdma_devices", "rdma_fabric",
"rdma_relaxed_ordering",
}
# Timing knobs, in the order the legacy colon-string encoded them, paired with the run_ep flag
# each one drives. The names match configs/sweep.json so a case's timing block is readable
Expand Down
8 changes: 8 additions & 0 deletions collectivex/runtime/prepare_backend.sh
Original file line number Diff line number Diff line change
Expand Up @@ -270,6 +270,14 @@ deepep_install() {
|| { collx_log "ERROR: DeepEP V2 environment activation failed"; return 1; }
collx_materialize_source "deepep-v2-$COLLX_DEEPEP_V2_COMMIT" "$source_dir" \
|| { collx_log "ERROR: DeepEP V2 staged source is invalid"; return 1; }
# Upstream registers the GIN window NCCL_WIN_STRICT_ORDERING, which drops PCIe relaxed ordering
# whatever NCCL_IB_PCI_RELAXED_ORDERING says; EP_WIN_RELAXED_ORDERING=1 opts a pool out. Unset,
# the registration is upstream's. A pin that moves or duplicates the flag fails here.
local window_cu="$source_dir/csrc/kernels/backend/nccl.cu"
local relaxed='get_env("EP_WIN_RELAXED_ORDERING", 0) ? NCCL_WIN_DEFAULT :'
[ "$(grep -c NCCL_WIN_STRICT_ORDERING "$window_cu")" = 1 ] \
&& sed -i "s/NCCL_WIN_STRICT_ORDERING/($relaxed &)/" "$window_cu" \
|| { collx_log "ERROR: DeepEP V2 window-ordering patch failed"; return 1; }
# The RDC device-link step (nvcc -dlink) gets no -gencode from the extension build, so nvcc
# falls back to its default arch (sm_75 on CUDA 13) and links kernels that cannot load on the
# target GPU (gb300/sm103: cudaErrorUnknown). NVCC_PREPEND_FLAGS reaches the dlink too.
Expand Down
14 changes: 14 additions & 0 deletions collectivex/tests/test_backends.py
Original file line number Diff line number Diff line change
Expand Up @@ -608,6 +608,20 @@ def test_flashinfer_and_nccl_ht_graph_decode_only(self):
self.assertTrue(_gate(nccl, "low-latency", phase="decode").cuda_graph_supported)


class GinWindowOrdering(unittest.TestCase):
"""Relaxed-ordering rows must not merge into the strict-ordered series they would replace."""

def test_only_relaxed_scale_out_rows_carry_the_suffix(self):
module = _import_stubbed("ep_deepep_v2", deep_ep=_deep_ep("ElasticBuffer", "Buffer"))
relaxed = {"EP_WIN_RELAXED_ORDERING": "1"}
for world_size, env, suffixed in ((16, relaxed, True), (16, {}, False), (8, relaxed, False)):
with self.subTest(world_size=world_size, env=env), \
mock.patch.dict(os.environ, env, clear=True):
backend = module.DeepEPV2Backend(
args(scale_up_domain=8, runner="h200-dgxc"), 0, world_size, 0, "cpu")
self.assertEqual(backend.kernel_generation.endswith("-relaxed-ordering"), suffixed)


class LowLatencyCapDecoupling(unittest.TestCase):
"""The LL receive size and the measured ladder must stay two numbers -- sizing the receive
from `max(ladder)` would shift every rung. Driven through the adapter with deep_ep stubbed,
Expand Down
14 changes: 14 additions & 0 deletions collectivex/tests/test_runtime.py
Original file line number Diff line number Diff line change
Expand Up @@ -245,6 +245,20 @@ def test_nvshmem_hca_list_per_placement(self) -> None:
self.assertEqual(stdout.strip().splitlines()[-1], expected)


class RelaxedOrderingProfileTests(unittest.TestCase):
def test_only_the_registry_flag_relaxes_the_deepep_window(self) -> None:
for flag, expected in (("1", "1"), ("0", "unset"), ("", "unset")):
with self.subTest(flag=flag):
stdout = run_common(
"export COLLX_RDMA_DEVICES=mlx5_0:1 EP_WIN_RELAXED_ORDERING=stale"
f" COLLX_RDMA_RELAXED_ORDERING='{flag}';"
" collx_apply_network_profile 2 nvlink-rdma;"
' echo "${EP_WIN_RELAXED_ORDERING:-unset}"',
check=True,
).stdout
self.assertEqual(stdout.strip().splitlines()[-1], expected)


class StageTests(unittest.TestCase):
def test_create_copy_and_validate_cleanup(self) -> None:
with tempfile.TemporaryDirectory() as directory:
Expand Down
Loading