diff --git a/AGENTS.md b/AGENTS.md index 4490a83e..f552902a 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -48,11 +48,11 @@ python3 -m venv .venv .venv/bin/python -m pytest ``` -Validate checked-in configurations through the production parser and common semantic checks -without initializing hardware: +The standard pre-PR check runs the portable tests, validates both retained and +generated configurations without initializing hardware, and checks the documentation: ```bash -python3 scripts/check_daqiri_configs.py --validator build/tools/daqiri_config_validate +scripts/check_pr.sh ``` The default pytest suite collects only `tests/portable/`. Build-backed C++ tests live under @@ -67,7 +67,7 @@ Integration and performance verification is done via the benchmark executables i | Executable | Source | Typical config | |---|---|---| -| `daqiri_bench_raw_gpudirect` | `raw_gpudirect_bench.cpp` | `daqiri_bench_raw_tx_rx.yaml`, `daqiri_bench_raw_tx_rx_4q.yaml`, `daqiri_bench_raw_tx_rx_spark.yaml`, `daqiri_bench_raw_{tx,rx}_spark_xhost.yaml`, `daqiri_bench_raw_sw_loopback.yaml`, `daqiri_bench_raw_hw_loopback_ibverbs.yaml`, `daqiri_bench_raw_rx_multi_q.yaml`, `daqiri_bench_raw_tx_rx_vxlan.yaml`, `daqiri_bench_raw_tx_rx_vlan.yaml`, `daqiri_bench_raw_tx_rx_gre.yaml`, `daqiri_bench_raw_tx_rx_nvgre.yaml`, `daqiri_bench_raw_tx_rx_spark_mq.yaml` (mq base; `run_spark_mq_bench.sh` derives the 4 cells via `scripts/gen_spark_mq_config.py`), `daqiri_bench_raw_tx_rx_pacing.yaml` (per-queue `pacing_mbps`; DPDK engine only) | +| `daqiri_bench_raw_gpudirect` | `raw_gpudirect_bench.cpp` | Canonical examples: `daqiri_bench_raw_tx_rx.yaml`, `daqiri_bench_raw_tx_rx_4q.yaml`, `daqiri_bench_raw_sw_loopback.yaml`, `daqiri_bench_raw_hw_loopback_ibverbs.yaml`, `daqiri_bench_raw_rx_multi_q.yaml`, `daqiri_bench_raw_tx_rx_pacing.yaml` (per-queue `pacing_mbps`; DPDK engine only). Generate Spark, cross-host, multi-queue matrix, and VLAN/VXLAN/GRE/NVGRE variants with `scripts/gen_daqiri_config.py`; the Spark harnesses invoke it directly. | | `daqiri_bench_raw_latency` | `raw_latency_bench.cpp` | `daqiri_bench_raw_latency_ibverbs.yaml` — caller-driven direct TX/RX, RX hardware timestamps, 64–8192-byte power-of-two latency sweep | | `daqiri_example_dynamic_rx_flow` | `dynamic_rx_flow_example.cpp` | `daqiri_example_dynamic_rx_flow.yaml` — `flow_isolation: true` startup followed by runtime scalar queue steering, multi-queue RSS, and raw-engine decap/pop flow add/delete | | `daqiri_example_dynamic_resource` | `dynamic_resource_example.cpp` | Any ibverbs config with at least one RX queue and one single-region TX queue (for example `daqiri_bench_raw_hw_loopback_ibverbs.yaml`) — initializes without RX queues, repeatedly adds/removes the first RX queue's steering flow, and exercises runtime MR and RX/TX queue add/delete | @@ -75,11 +75,11 @@ Integration and performance verification is done via the benchmark executables i | `daqiri_bench_raw_hds` | `raw_hds_bench.cpp` | `daqiri_bench_raw_tx_rx_hds.yaml` | | `daqiri_bench_raw_reorder_seq` | `raw_reorder_seq_bench.cpp` | `daqiri_bench_raw_tx_rx_reorder_seq_1024*.yaml`, `daqiri_bench_raw_rx_reorder_seq_*.yaml` | | `daqiri_bench_raw_reorder_quantize` | `raw_reorder_quantize_bench.cpp` | `daqiri_bench_raw_tx_rx_reorder_quantize_seq_batch.yaml` | -| `daqiri_bench_rdma` | `rdma_bench.cpp` | `daqiri_bench_rdma_tx_rx.yaml`, `daqiri_bench_rdma_tx_rx_spark.yaml`, `daqiri_bench_rdma_tx_rx_spark_xhost.yaml`, `daqiri_bench_rdma_tx_rx_spark_netns.yaml` (combined-role netns base; `run_spark_bench.sh` splits per role via `scripts/gen_spark_netns_config.py`) | -| `daqiri_bench_socket` | `socket_bench.cpp` | `daqiri_bench_socket_{udp,tcp}_tx_rx.yaml`, `daqiri_bench_socket_{udp,tcp}_tx_rx_spark_netns.yaml` (combined-role netns bases), `daqiri_bench_socket_{udp,tcp}_{client,server}_spark_xhost.yaml` (cross-host role configs) | +| `daqiri_bench_rdma` | `rdma_bench.cpp` | Canonical example: `daqiri_bench_rdma_tx_rx.yaml`. Generate Spark netns/cross-host client and server roles with `scripts/gen_daqiri_config.py socket-pair --transport roce`. | +| `daqiri_bench_socket` | `socket_bench.cpp` | Canonical examples: `daqiri_bench_socket_{udp,tcp}_tx_rx.yaml`. Generate Spark netns/cross-host client and server roles with `scripts/gen_daqiri_config.py socket-pair`. | | `daqiri_pool_ring_bench` | `pool_ring_bench.cpp` | none — microbenchmark comparing `daqiri::Ring`/`daqiri::ObjectPool` vs DPDK `rte_ring`/`rte_mempool` (SPSC/MPMC, single/bulk, thread sweep). The `rte_*` comparison arm compiles only in a DPDK-enabled build; takes no YAML/CLI args | -The four `raw_*` benches share `raw_bench_common.{cpp,h}` and accept `--seconds N`. `daqiri_bench_rdma` and `daqiri_bench_socket` also take `--mode {tx,rx,both}`. `daqiri_bench_raw_gpudirect`, `daqiri_bench_raw_hds`, `daqiri_bench_rdma`, and `daqiri_bench_socket` additionally accept `--workload none|fft|gemm|gemm_fp16` — a reusable representative GPU workload (`examples/bench_workload.{h,cu}`, cuFFT/cuBLAS) run once per received reorder window on the **actual received payload**. Each backend first assembles the burst's payloads into one contiguous GPU buffer via `examples/bench_pipeline.{h,cu}` (`ReorderPipeline`): a sequence-number reorder kernel for the out-of-order transports (DPDK raw, UDP) and an arrival-order gather for the in-order ones (RoCE RC, TCP); sockets stage host→device first since their payloads land in pageable host memory. The reorder/gather kernels (`packet_reorder_copy_payload_by_sequence`, `packet_gather_copy_payload`) live in `src/kernels.cu`. `gemm` is FP32 `cublasSgemm`; `gemm_fp16` is the same-size mixed-precision FP16/tensor-core `cublasGemmEx` (inference-style); the contiguous buffer supplies the FFT input / GEMM A operand. `--workload-gemm-dim N` pins the square GEMM side length (default 1024), so the FLOP count per call (2·n³) is FIXED and the compute working set is exactly n·n·elem_size, read from the front of each received I/O unit (the unit must be at least that large). `--workload-fft-len N` pins the 1-D C2C transform length for `fft` (default 1024; independent of the GEMM dimension); the working set is fanned out across as many batched length-N transforms as fit. Used by `run_spark_bench.sh`'s `WORKLOAD` / `GEMM_DIM` / `FFT_LEN` env (all backends) to fill the CSV `post_process` / `post_process_gemm_dim` columns (issue #15). +The four `raw_*` benches share `raw_bench_common.{cpp,h}` and accept `--seconds N`. `daqiri_bench_rdma` and `daqiri_bench_socket` also take `--mode {server,client,both}`. `daqiri_bench_raw_gpudirect`, `daqiri_bench_raw_hds`, `daqiri_bench_rdma`, and `daqiri_bench_socket` additionally accept `--workload none|fft|gemm|gemm_fp16` — a reusable representative GPU workload (`examples/bench_workload.{h,cu}`, cuFFT/cuBLAS) run once per received reorder window on the **actual received payload**. Each backend first assembles the burst's payloads into one contiguous GPU buffer via `examples/bench_pipeline.{h,cu}` (`ReorderPipeline`): a sequence-number reorder kernel for the out-of-order transports (DPDK raw, UDP) and an arrival-order gather for the in-order ones (RoCE RC, TCP); sockets stage host→device first since their payloads land in pageable host memory. The reorder/gather kernels (`packet_reorder_copy_payload_by_sequence`, `packet_gather_copy_payload`) live in `src/kernels.cu`. `gemm` is FP32 `cublasSgemm`; `gemm_fp16` is the same-size mixed-precision FP16/tensor-core `cublasGemmEx` (inference-style); the contiguous buffer supplies the FFT input / GEMM A operand. `--workload-gemm-dim N` pins the square GEMM side length (default 1024), so the FLOP count per call (2·n³) is FIXED and the compute working set is exactly n·n·elem_size, read from the front of each received I/O unit (the unit must be at least that large). `--workload-fft-len N` pins the 1-D C2C transform length for `fft` (default 1024; independent of the GEMM dimension); the working set is fanned out across as many batched length-N transforms as fit. Used by `run_spark_bench.sh`'s `WORKLOAD` / `GEMM_DIM` / `FFT_LEN` env (all backends) to fill the CSV `post_process` / `post_process_gemm_dim` columns (issue #15). ```bash ./build/examples/daqiri_bench_raw_gpudirect ./build/examples/daqiri_bench_raw_tx_rx.yaml --seconds 10 @@ -174,7 +174,7 @@ The web docs live in `docs/` and are built with [MkDocs Material](https://squidf - `docs/getting-started.md` — system requirements, build instructions, and first benchmark smoke-test guidance. Only add information to Getting Started when it directly affects requirements, library build steps, or benchmark smoke-test instructions. - `docs/concepts.md` — terminology glossary (stream types and endpoint URI schemes, GPUDirect, packet/burst/segment, flow/queue, memory region, zero-copy ownership, RX reorder). Meant to be opened in parallel with the rest of the docs. - `docs/api-reference/index.md` — API guide (6-step application lifecycle, configuration-first model) -- `docs/api-reference/configuration.md`, `docs/api-reference/cpp.md`, `docs/api-reference/python.md` — YAML schema, C++ API, and Python bindings docs +- `docs/api-reference/configuration.md`, `docs/api-reference/cpp.md`, `docs/api-reference/python.md` — YAML reference, C++ API, and Python bindings docs - `docs/tutorials/` — tutorial walkthroughs (system config, config-file walkthrough, Holoscan integration, ResNet inference) - `docs/benchmarks/` — benchmark guide pages, surfaced as a top-level "Benchmarking" nav section in `mkdocs.yml` and the landing page (`docs/index.md`): - `docs/benchmarks/index.md` — overview and engine-selection decision tree @@ -183,7 +183,7 @@ The web docs live in `docs/` and are built with [MkDocs Material](https://squidf - `docs/benchmarks/performance-dgx-spark.md` — per-platform performance report for DGX Spark stream/protocol combinations (the long internal report lives outside the repo in `projects/daqiri-notes/`) - `docs/stylesheets/extra.css` — custom theme overrides -**User-facing vocabulary:** the YAML schema uses `stream_type` (`raw`, `socket`, future `pcie`); for socket streams the transport is encoded in the endpoint URI scheme (`udp://`, `tcp://`, `roce://`) in `socket_config.local_addr`/`remote_addr`, **not** a separate `protocol` field. (`SocketProtocol` still exists internally, derived from the scheme.) **"Engine"** is the standard term for the specific library backing an implementation; it replaced the former "manager" and "backend" terms and is now used consistently across code (`src/engines//`, the `Engine` ABC, CMake `DAQIRI_ENGINE`), the API reference, tutorials, the landing page, and concept pages. The mapping: `stream_type: "raw"` is implemented by the `dpdk` engine; `stream_type: "socket"` with `udp://`/`tcp://` endpoints by the always-built `socket` engine; `stream_type: "socket"` with `roce://` endpoints by the `ibverbs` engine. +**User-facing vocabulary:** the YAML format uses `stream_type` (`raw`, `socket`, future `pcie`); for socket streams the transport is encoded in the endpoint URI scheme (`udp://`, `tcp://`, `roce://`) in `socket_config.local_addr`/`remote_addr`, **not** a separate `protocol` field. (`SocketProtocol` still exists internally, derived from the scheme.) **"Engine"** is the standard term for the specific library backing an implementation; it replaced the former "manager" and "backend" terms and is now used consistently across code (`src/engines//`, the `Engine` ABC, CMake `DAQIRI_ENGINE`), the API reference, tutorials, the landing page, and concept pages. The mapping: `stream_type: "raw"` is implemented by the `dpdk` engine; `stream_type: "socket"` with `udp://`/`tcp://` endpoints by the always-built `socket` engine; `stream_type: "socket"` with `roce://` endpoints by the `ibverbs` engine. **Keeping docs in sync with code:** before committing changes, scan for the recurring drift hotspots: - **Stream-type list** (`src/engines/*/`) — README Engines table, `docs/getting-started.md`, `docs/concepts.md` (Stream Types section + Support and testing admonition), `docs/api-reference/configuration.md` diff --git a/CMakeLists.txt b/CMakeLists.txt index 4b589f0e..17966e7c 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -82,6 +82,27 @@ if(DAQIRI_BUILD_APPLICATIONS) add_subdirectory(applications) endif() +set(DAQIRI_CONFIG_GENERATOR_DATA_SUBDIR "daqiri/config-generator") +if(IS_ABSOLUTE "${CMAKE_INSTALL_DATADIR}") + set(DAQIRI_CONFIG_GENERATOR_MODULE_HINT + "${CMAKE_INSTALL_DATADIR}/${DAQIRI_CONFIG_GENERATOR_DATA_SUBDIR}") +elseif(IS_ABSOLUTE "${CMAKE_INSTALL_BINDIR}") + set(DAQIRI_CONFIG_GENERATOR_MODULE_HINT + "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_DATADIR}/${DAQIRI_CONFIG_GENERATOR_DATA_SUBDIR}") +else() + file(RELATIVE_PATH DAQIRI_CONFIG_GENERATOR_MODULE_HINT + "/${CMAKE_INSTALL_BINDIR}" + "/${CMAKE_INSTALL_DATADIR}/${DAQIRI_CONFIG_GENERATOR_DATA_SUBDIR}") +endif() +configure_file(scripts/gen_daqiri_config.py + ${CMAKE_CURRENT_BINARY_DIR}/gen_daqiri_config.py + @ONLY) +install(PROGRAMS ${CMAKE_CURRENT_BINARY_DIR}/gen_daqiri_config.py + DESTINATION ${CMAKE_INSTALL_BINDIR}) +install(DIRECTORY scripts/daqiri_config + DESTINATION ${CMAKE_INSTALL_DATADIR}/${DAQIRI_CONFIG_GENERATOR_DATA_SUBDIR} + PATTERN "__pycache__" EXCLUDE) + install( EXPORT daqiriTargets NAMESPACE daqiri:: diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f648ed49..e17660f9 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -89,8 +89,8 @@ runners. See `tests/README.md` for the container dependency command, supported invocations, and marker policy. Build `daqiri_config_validate` in the required project container before running -`scripts/check_pr.sh`. The check script validates representative checked-in configurations -through the production C++ parser and hardware-independent semantic checks. Set +`scripts/check_pr.sh`. The check script validates the retained and generated configuration +matrices through the production C++ parser and hardware-independent semantic checks. Set `DAQIRI_CONFIG_VALIDATOR` when the executable is not at `build/tools/daqiri_config_validate`. When a new example config exercises a configuration form these cases do not cover, add a diff --git a/Dockerfile b/Dockerfile index 17d6f9a5..1bb0cce3 100644 --- a/Dockerfile +++ b/Dockerfile @@ -323,6 +323,8 @@ RUN cmake -S . -B build \ && cmake --build build -j "$(nproc)" \ && python3 scripts/check_daqiri_configs.py \ --validator build/tools/daqiri_config_validate \ + && python3 scripts/check_generated_configs.py \ + --validator build/tools/daqiri_config_validate \ && cmake --install build # ============================== diff --git a/README.md b/README.md index eeaf65a5..1c2718ba 100644 --- a/README.md +++ b/README.md @@ -93,6 +93,7 @@ Reference material for the DAQIRI codebase: - [Concepts](https://nvidia.github.io/daqiri/concepts/) — Glossary of DAQIRI terminology (kernel bypass, GPUDirect, packet/burst/segment, flow/queue, memory region, zero-copy ownership, RX reorder). Meant to be opened in parallel with the rest of the docs. - [API Guide](https://nvidia.github.io/daqiri/api-reference/) — Six-step DAQIRI application lifecycle and configuration-first model - [Configuration YAML Reference](https://nvidia.github.io/daqiri/api-reference/configuration/) — Full YAML config reference for all engines +- [Configuration Generation](https://nvidia.github.io/daqiri/config-generation/) — Deterministic production, benchmark, multi-queue, and cross-host configurations - [C++ API Usage](https://nvidia.github.io/daqiri/api-reference/cpp/) — C++ RX/TX workflows, buffer lifecycle, file writing, utilities, and status codes - [Python API Usage](https://nvidia.github.io/daqiri/api-reference/python/) — Python bindings, workflow examples, enums, config classes, and helper functions - [Performance: DGX Spark](https://nvidia.github.io/daqiri/benchmarks/performance-dgx-spark/) — Per-platform throughput, drop, and utilization numbers for stream/protocol combinations on DGX Spark diff --git a/docs/api-reference/configuration.md b/docs/api-reference/configuration.md index 54d665f4..e869f947 100644 --- a/docs/api-reference/configuration.md +++ b/docs/api-reference/configuration.md @@ -10,9 +10,13 @@ Either form defines memory regions, NIC interfaces, TX/RX queues, and flow rules is passed to `daqiri_init()` at startup. The struct form is useful for customers who want to interoperate with existing configuration code. -See `examples/daqiri_bench_*.yaml` for complete working examples. +Start with the commented configurations under `examples/` to see complete, +readable configurations and understand how the fields fit together. When you +need repeatable production, benchmark, multi-queue, or cross-host variants, use +[Configuration Generation](../config-generation.md) to apply system and topology +parameters consistently. -## Validate without hardware initialization +## Optional validation without hardware initialization Use `daqiri_config_validate` to check YAML files before running an application, such as in CI or on a machine without the target NIC. `daqiri_init()` performs these checks during startup. The @@ -24,8 +28,9 @@ daqiri_config_validate config.yaml another-config.yaml ``` The command exits with status `0` when every file is valid, `1` when any file is invalid, and -`2` when no file was provided. It is built and installed even when -`DAQIRI_BUILD_EXAMPLES=OFF`. +`2` when no file was provided. Use `daqiri_config_validate --list-engines` to print the +engines compiled into the validator and exit successfully without checking files. +It is built and installed even when `DAQIRI_BUILD_EXAMPLES=OFF`. OpenTelemetry metrics do not add YAML fields. Metrics-enabled builds use the same interface, queue, and flow names from the active configuration as metric @@ -55,6 +60,7 @@ These settings apply globally to both TX and RX: - **`log_level`**: Engine log level. - type: `string` - values: `trace`, `debug`, `info`, `warn` (default), `error`, `critical`, `off` + - any other value is rejected during configuration parsing - **`loopback`**: Select a loopback mode for local testing. - type: `string` - values: `""` (disabled, default), `"sw"` (DPDK software loopback, no NIC), @@ -406,7 +412,7 @@ weights. RSS is flow-affine: every packet with an unchanged five tuple stays on one queue. Roughly even packet counts require enough distinct tuples with reasonably balanced traffic; this is not packet striping or exact round-robin delivery. -There is no queue-action mode field in schema v1; a future stripe mode can be +There is no queue-action mode field in configuration version 1; a future stripe mode can be added without changing the multi-ID RSS default. If the NIC rejects an RSS action, static initialization or the dynamic flow completion fails rather than falling back to one queue. diff --git a/docs/api-reference/index.md b/docs/api-reference/index.md index c5cf07c3..6a48b879 100644 --- a/docs/api-reference/index.md +++ b/docs/api-reference/index.md @@ -41,7 +41,7 @@ endpoints after initialization, resolve their opaque `EndpointId` handles, and select a configured TX queue for each submission. These endpoints provide payload-only TX buffers and are separate from socket endpoint URIs. -The configuration schema lives in the +The configuration format is documented in the [Configuration YAML Reference](configuration.md). For an annotated end-to-end example, see the [configuration walkthrough tutorial](../tutorials/configuration-walkthrough.md). diff --git a/docs/benchmarks/raw_benchmarking.md b/docs/benchmarks/raw_benchmarking.md index 88c1f379..12e9ef97 100644 --- a/docs/benchmarks/raw_benchmarking.md +++ b/docs/benchmarks/raw_benchmarking.md @@ -68,20 +68,15 @@ docker run --rm -it --privileged \ !!! tip "DGX Spark" - For systems configured per the [DGX Spark profile](../tutorials/system_configuration.md#dgx-spark-profile), use these configs to skip the PCIe/IP/CPU-core edits below: + Start with [`daqiri_bench_raw_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx.yaml) to see how the TX and RX configuration fits together. On a system configured per the [DGX Spark profile](../tutorials/system_configuration.md#dgx-spark-profile), adapt the NIC addresses, use `host_pinned` memory, place the queue and application workers on the high-frequency cores, and set the destination MAC to the receiving port. [Generate a raw-Ethernet pair](../config-generation.md#generate-a-raw-ethernet-pair) can apply those system parameters and produce the concrete loopback. The `rx_port` is `0002:01:00.1` (physical port p1), so read its MAC with `cat /sys/class/net/enP2p1s0f1np1/address`. - - [`daqiri_bench_raw_tx_rx_spark.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_spark.yaml) for `daqiri_bench_raw_gpudirect`. Still set `eth_dst_addr` to the RX MAC. The rx_port is `0002:01:00.1` (physical port p1), so read its MAC: `cat /sys/class/net/enP2p1s0f1np1/address`. This p0-to-p1 pairing is intentional for an over-the-wire single-machine loopback. Using two PFs that map to the same physical port exercises the on-chip eswitch path instead. - - [`daqiri_bench_rdma_tx_rx_spark.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_rdma_tx_rx_spark.yaml) for `daqiri_bench_rdma`. No further edits needed. - - For the multi-queue core-scaling matrix, use the single base config [`daqiri_bench_raw_tx_rx_spark_mq.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_spark_mq.yaml) (the balanced TX=2/RX=2 superset) with `daqiri_bench_raw_gpudirect`, driven by [`run_spark_mq_bench.sh`](https://github.com/nvidia/daqiri/blob/main/examples/run_spark_mq_bench.sh). It derives the four `(TX, RX)` cells from the base (via `scripts/gen_spark_mq_config.py`) and sweeps the payload. - - The Spark configs also pin the benchmark application's `bench_tx.cpu_core` / `bench_rx.cpu_core` fields to the high-frequency Cortex-X925 cores. Keep both the DAQIRI queue cores and the application worker cores on cores 16-19 unless you intentionally want a lower-power core in the measurement. + For multiple queues, [`daqiri_bench_raw_tx_rx_4q.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_4q.yaml) shows how queues, memory regions, flow steering, and application workers relate. [`run_spark_bench.sh`](https://github.com/nvidia/daqiri/blob/main/examples/run_spark_bench.sh) generates each single-queue benchmark cell, while [`run_spark_mq_bench.sh`](https://github.com/nvidia/daqiri/blob/main/examples/run_spark_mq_bench.sh) generates all four `(TX, RX)` multi-queue combinations. #### Cross-host two-DGX-Spark loopback -If you have two DGX Sparks cross-cabled p0↔p0 instead of a chassis QSFP loop on one machine, use the `_xhost` configs. Each host runs only its own role, so the YAML on each side configures one port instead of two. Both hosts must already be set up per the [DGX Spark profile](../tutorials/system_configuration.md#dgx-spark-profile), with one adjustment: the `daqiri-tx` (`1.1.1.1/24`) and `daqiri-rx` (`2.2.2.2/24`) nmcli profiles are *split across* the two hosts. Bring up `daqiri-tx` on the TX host's p0 and `daqiri-rx` on the RX host's p0, instead of both on one box. +If you have two DGX Sparks cross-cabled p0↔p0 instead of a chassis QSFP loop on one machine, generate independent TX and RX files with `raw-pair --role both`. Each host runs only its own role, so the YAML on each side configures one port instead of two. Both hosts must already be set up per the [DGX Spark profile](../tutorials/system_configuration.md#dgx-spark-profile), with one adjustment: the `daqiri-tx` (`1.1.1.1/24`) and `daqiri-rx` (`2.2.2.2/24`) nmcli profiles are *split across* the two hosts. Bring up `daqiri-tx` on the TX host's p0 and `daqiri-rx` on the RX host's p0, instead of both on one box. -**Network prerequisite.** Assigning `/24` addresses on each host is not enough for the kernel to reach the peer over a direct cable. Install a host route on the cabled port by running [`scripts/setup_spark_xhost_net.sh`](https://github.com/nvidia/daqiri/blob/main/scripts/setup_spark_xhost_net.sh) on **both** hosts after bringing up the nmcli profile. The ibverbs raw engine uses that route and Linux ARP when `eth_dst_addr` is omitted; RDMA-CM uses the same kernel route. See the [cross-host variant](../tutorials/system_configuration.md#cross-host-variant-two-sparks) in System Configuration for the full steps. +**Network setup.** Assigning `/24` addresses on each host is not enough for the kernel to reach the peer over a direct cable. Install a host route on the cabled port by running [`scripts/setup_spark_xhost_net.sh`](https://github.com/nvidia/daqiri/blob/main/scripts/setup_spark_xhost_net.sh) on **both** hosts after bringing up the nmcli profile. RDMA-CM uses that route; raw ibverbs can also use it with Linux ARP when `eth_dst_addr` is omitted. The generated raw pair below supplies the RX port MAC explicitly. See the [cross-host variant](../tutorials/system_configuration.md#cross-host-variant-two-sparks) in System Configuration for the full steps. ```bash # TX host @@ -99,10 +94,10 @@ ip route get # must name enp1s0f0np0, not lo ```bash # RX host -sudo ./daqiri_bench_raw_gpudirect daqiri_bench_raw_rx_spark_xhost.yaml --seconds 30 +sudo ./daqiri_bench_raw_gpudirect generated/raw-xhost/rx.yaml --seconds 30 -# TX host (the ibverbs config resolves the RX MAC through Linux ARP) -sudo ./daqiri_bench_raw_gpudirect daqiri_bench_raw_tx_spark_xhost.yaml --seconds 30 +# TX host; supply the RX port MAC when generating this file +sudo ./daqiri_bench_raw_gpudirect generated/raw-xhost/tx.yaml --seconds 30 ``` Verify both sides report non-zero packet counts and no `NO_FREE_BURST_BUFFERS` / `NO_FREE_PACKET_BUFFERS` errors. @@ -119,10 +114,10 @@ intended queue. ```bash # RX (server) host -sudo ./daqiri_bench_rdma daqiri_bench_rdma_tx_rx_spark_xhost.yaml --mode server --seconds 30 +sudo ./daqiri_bench_rdma generated/roce/rx.yaml --mode server --seconds 30 # TX (client) host -sudo ./daqiri_bench_rdma daqiri_bench_rdma_tx_rx_spark_xhost.yaml --mode client --seconds 30 +sudo ./daqiri_bench_rdma generated/roce/tx.yaml --mode client --seconds 30 ``` Verify both sides report non-zero send/receive completions and no `CQ error` / `RETRY_EXC_ERR` lines in the client log. @@ -216,12 +211,10 @@ encapsulate on TX and pop or decapsulate on RX. The application packet buffers remain pre-encap on TX and post-decap on RX; DAQIRI accounts for outer-header overhead when sizing MTU/wire frames. -| Transform | YAML config | Binary | -|---|---|---| -| VXLAN encap + decap | [`daqiri_bench_raw_tx_rx_vxlan.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_vxlan.yaml) | `daqiri_bench_raw_gpudirect` | -| VLAN push + pop | [`daqiri_bench_raw_tx_rx_vlan.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_vlan.yaml) | `daqiri_bench_raw_gpudirect` | -| GRE encap + decap | [`daqiri_bench_raw_tx_rx_gre.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_gre.yaml) | `daqiri_bench_raw_gpudirect` | -| NVGRE encap + decap | [`daqiri_bench_raw_tx_rx_nvgre.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_nvgre.yaml) | `daqiri_bench_raw_gpudirect` | +Generate each variant with the same raw-pair command and one of +`--transform vlan`, `--transform vxlan`, `--transform gre`, or +`--transform nvgre`; see [configuration generation](../config-generation.md#generate-a-raw-ethernet-pair). +All four outputs run on `daqiri_bench_raw_gpudirect`. ##### Identify your NIC's PCIe addresses @@ -314,7 +307,7 @@ bench_tx: - The benchmark reuses the resolved MAC for the entire run; it does not monitor Linux neighbor changes. Long-running applications should define their own refresh policy and call `resolve_ipv4_mac()` again when a peer, gateway, route, link, or namespace may have changed. See [Destination MAC resolution](../concepts.md#destination-mac-resolution). - `cpu_core` - the benchmark application's own TX worker thread affinity. Set the matching `bench_rx.cpu_core` for RX workers too. These app-thread fields are distinct from the DAQIRI queue `cpu_core` values that poll the NIC. - We ignore the IP fields (`ip_src_addr`, `ip_dst_addr`) for now, as we are testing on a layer 2 network by just connecting a cable between the two interfaces on our system, therefore having mock values has no impact. - - You might have noted the lack of a `eth_src_addr` field in this `bench_tx` section. This is because the source Ethernet MAC address can be inferred automatically by the DAQIRI library from the PCIe address of the Tx interface referenced above. + - You might have noted the lack of an `eth_src_addr` field in this `bench_tx` section. The DPDK `tx_eth_src` egress flow rewrites the source MAC from the TX interface. For an ibverbs benchmark profile, supply the TX port MAC as `eth_src_addr` because the benchmark copies a complete packet-header template into its buffers. ## Run the loopback test diff --git a/docs/benchmarks/socket_benchmarking.md b/docs/benchmarks/socket_benchmarking.md index 2c2916d7..515261b6 100644 --- a/docs/benchmarks/socket_benchmarking.md +++ b/docs/benchmarks/socket_benchmarking.md @@ -7,7 +7,7 @@ hide: Use this page when the peer transport is TCP, UDP, or RoCE/RDMA. These benchmarks use the Linux networking stack for TCP/UDP and RDMA verbs for RoCE, so the same client/server namespace shape is useful for proving that traffic leaves the host through the expected NIC path. -For **two-host Spark cross-cable** tests (not netns), RoCE/RDMA still needs kernel reachability to the peer, so apply the host route and static neighbor steps in [System Configuration → Cross-host variant](../tutorials/system_configuration.md#cross-host-variant-two-sparks) before running `daqiri_bench_rdma` with the `_xhost` configs. +For **two-host Spark cross-cable** tests (not netns), RoCE/RDMA needs kernel reachability to the peer. Set up the host routes on both machines as shown in [System Configuration → Cross-host variant](../tutorials/system_configuration.md#cross-host-variant-two-sparks) before running `daqiri_bench_rdma` with the generated client and server configs. The setup script permits dynamic ARP or an optional static neighbor. Make sure to [build DAQIRI](../getting-started.md#build-the-daqiri-library) with the `ibverbs` engine first (for the RoCE/RDMA benchmark); Linux UDP/TCP sockets are always available. @@ -228,7 +228,11 @@ after processing it. TCP `rx.queues[].batch_size` does not change this bound; the socket TCP path surfaces one packet per `recv()` chunk. UDP retains its existing datagram receive behavior and does not wait for this TCP queue capacity. -For an on-wire namespace test, use separate server and client YAML files. The important fields are the endpoint URI scheme, namespace IPs, server port, `max_payload_size`, memory-region `buf_size`, and benchmark `message_size`. +For an on-wire namespace test, generate separate server and client YAML files +with [`gen_daqiri_config.py socket-pair`](../config-generation.md#generate-udp-tcp-or-roce-roles). +Check the endpoint URI scheme, namespace IPs, server port, `max_payload_size`, +memory-region `buf_size`, and benchmark `message_size` in the generated files +before running. For UDP, `rx.queues[].cpu_core` pins the DAQIRI socket I/O thread that drains `recvmmsg()`. The separate `socket_bench_*.cpu_core` pins the application worker @@ -378,11 +382,13 @@ For a four-process run, create four server/client YAML pairs with unique server ## Run the RDMA RoCE benchmark -Start from `examples/daqiri_bench_rdma_tx_rx.yaml` or `examples/daqiri_bench_rdma_tx_rx_spark.yaml`. The full config can run both endpoints in one process: +Start from the canonical `examples/daqiri_bench_rdma_tx_rx.yaml` for a one-process +smoke test, or generate independent roles for namespace/cross-host runs. The full +canonical config can run both endpoints in one process: ```bash ./build-socket-rdma/examples/daqiri_bench_rdma \ - examples/daqiri_bench_rdma_tx_rx_spark.yaml \ + examples/daqiri_bench_rdma_tx_rx.yaml \ --seconds 10 --mode both ``` diff --git a/docs/config-generation.md b/docs/config-generation.md new file mode 100644 index 00000000..de0a9c0e --- /dev/null +++ b/docs/config-generation.md @@ -0,0 +1,279 @@ +# Generate Configurations + +Start with the commented YAML files under `examples/` to understand how a +complete DAQIRI configuration fits together. `scripts/gen_daqiri_config.py` +then provides a deterministic path from deployment parameters to DAQIRI YAML. +It preserves application-owned top-level sections and writes byte-identical +output for identical inputs. + +The generator requires Python 3 and PyYAML. They are installed in the DAQIRI +development container. For a host Python environment: + +```bash +python3 -m pip install pyyaml +``` + +A CMake install exposes the same entry point as +`/opt/daqiri/bin/gen_daqiri_config.py`; commands below use the source-tree path. + +## Generate a raw-Ethernet pair + +Supply the actual local topology rather than editing a copied YAML. After +identifying the physical TX and RX netdevs, set `TX_IF` and `RX_IF` to their +names. Read the PCI addresses and both port MAC addresses from sysfs: + +```bash +TX_PCI="$(basename "$(readlink -f "/sys/class/net/$TX_IF/device")")" +RX_PCI="$(basename "$(readlink -f "/sys/class/net/$RX_IF/device")")" +TX_MAC="$(cat "/sys/class/net/$TX_IF/address")" +RX_MAC="$(cat "/sys/class/net/$RX_IF/address")" +printf 'TX_PCI=%s\nRX_PCI=%s\nTX_MAC=%s\nRX_MAC=%s\n' \ + "$TX_PCI" "$RX_PCI" "$TX_MAC" "$RX_MAC" +``` + +Example output with invented addresses (use the values printed on your system): + +```text +TX_PCI=0000:aa:00.0 +RX_PCI=0000:aa:00.1 +TX_MAC=02:00:00:00:00:01 +RX_MAC=02:00:00:00:00:02 +``` + +Pass the full values to this IGX-style loopback example: + +```bash +python3 scripts/gen_daqiri_config.py raw-pair \ + --tx-address "$TX_PCI" --rx-address "$RX_PCI" \ + --master-core 3 --engine ibverbs --memory-kind device \ + --tx-queue-cores 4 --rx-queue-cores 5 \ + --tx-worker-cores 6 --rx-worker-cores 7 \ + --eth-src-addr "$TX_MAC" --eth-dst-addr "$RX_MAC" \ + --output raw-loopback.yaml +``` + +`--role loopback` is the default and emits one document containing TX and RX. +For two hosts, run the lookup on each host. Set `TX_PCI` and `RX_PCI` to the +ports on their respective hosts, `TX_MAC` to the transmitting host's port MAC, +and `RX_MAC` to the receiving host's port MAC. Then generate one independently +runnable file per role: + +```bash +python3 scripts/gen_daqiri_config.py raw-pair \ + --tx-address "$TX_PCI" --rx-address "$RX_PCI" \ + --master-core 8 --engine dpdk --memory-kind host_pinned \ + --tx-queue-cores 17 --rx-queue-cores 18 \ + --tx-worker-cores 16 --rx-worker-cores 19 \ + --eth-dst-addr "$RX_MAC" \ + --role both --output-dir generated/raw-xhost +``` + +The TX file contains only the TX interface, TX memory, and `bench_tx`; the RX +file contains only the RX equivalents. Generate from the local and peer system +facts, copy the file for the remote role to that host, start RX first, and then +start TX. File transfer and remote process control intentionally remain outside +the deterministic generator. + +Comma-separated core lists create multi-queue matrices. For example, +`--tx-queue-cores 16,19 --tx-worker-cores 15,6` creates two TX queues. The +generator derives memory regions, flow-to-queue routing, and benchmark entries +from the queue counts. `examples/run_spark_mq_bench.sh` uses this interface for +all four 1×1, 1×2, 2×1, and 2×2 cells. + +Raw memory-region `buf_size` defaults to `header_size + payload_size`. Set +`--buffer-size` (also accepted as `--buf-size`) to keep packet-buffer capacity +fixed across a payload sweep; both Spark DPDK harnesses use `8064` bytes to +preserve their published methodology. + +When `--batch-size` is omitted, DPDK profiles use `10240` packets per burst and +ibverbs or engine-default profiles use `1024`. The ibverbs engine also checks +the NIC's `max_qp_wr` during initialization and permits at most half that value; +pass a smaller explicit batch size if the runtime reports a lower limit. + +Add one of `--transform vlan`, `--transform vxlan`, `--transform gre`, or +`--transform nvgre` to generate the corresponding raw hardware encap/decap +configuration. Transform profiles currently require one TX and one RX queue. +They also require an explicit `--engine dpdk` or `--engine ibverbs`, because the +two engines express the transform's RX match differently. + +Use `--daqiri-only` when generating a production library config rather than a +benchmark input. This omits `bench_tx` and `bench_rx`. Raw TX queues enable the +optional `tx_eth_src` offload by default; pass `--no-tx-eth-src` when the +application supplies the Ethernet source address itself or the NIC cannot +program that offload. + +Benchmark-owned ibverbs profiles require `--eth-src-addr` because the raw +benchmark copies a complete Ethernet/IP/UDP template directly into each packet +buffer. DPDK benchmark profiles may omit it because the DPDK egress flow +rewrites the source MAC. Applications generated with `--daqiri-only` may omit +it and use DAQIRI's header helpers with `tx_eth_src` instead. + +## Generate UDP, TCP, or RoCE roles + +`socket-pair` emits separate TX/client and RX/server documents. The same command +works with `--transport udp`, `tcp`, or `roce`. + +Transport-specific options are checked rather than ignored: `--rx-batch-size` +and `--iterations` apply only to TCP/UDP, while `--rx-num-bufs`, +`--tx-num-bufs`, `--rx-depth`, `--tx-depth`, and `--roce-transport-mode` apply +only to RoCE. + +Set `CLIENT_IP` and `SERVER_IP` to the addresses assigned to the two endpoints +before generating this UDP namespace example: + +```bash +python3 scripts/gen_daqiri_config.py socket-pair \ + --transport udp \ + --client-address "$CLIENT_IP" --server-address "$SERVER_IP" \ + --client-port 5101 --server-port 5001 \ + --client-master-core 8 --server-master-core 8 \ + --client-rx-core 17 --client-tx-core 17 \ + --server-rx-core 16 --server-tx-core 16 \ + --client-worker-core 17 --server-worker-core 16 \ + --message-size 8000 --buffer-size 65536 \ + --num-bufs 1024 --rx-batch-size 32 \ + --role both --output-dir generated/udp +``` + +For the two-host Spark RoCE setup, set `TX_HOST_IP` and `RX_HOST_IP` to the +addresses assigned to the `daqiri-tx` and `daqiri-rx` profiles in the +[system configuration tutorial](tutorials/system_configuration.md#cross-host-variant-two-sparks). +Read each profile on its host: + +```bash +# TX host +nmcli -g ipv4.addresses connection show daqiri-tx +# RX host +nmcli -g ipv4.addresses connection show daqiri-rx +``` + +Example output with the final octets obscured is `1.1.1.x/24` on TX and +`2.2.2.x/24` on RX. Use the complete addresses without `/24` for `TX_HOST_IP` +and `RX_HOST_IP`. The generated `tx.yaml` uses the TX address as its local +endpoint; `rx.yaml` uses the RX address. + +Set up the host route on both hosts as shown in that tutorial before running +the benchmark. Without a peer MAC, the setup script leaves neighbor resolution +to Linux ARP; supplying one installs a static neighbor. Use host-pinned memory +and size receive/transmit windows explicitly when needed: + +```bash +python3 scripts/gen_daqiri_config.py socket-pair \ + --transport roce \ + --client-address "$TX_HOST_IP" --server-address "$RX_HOST_IP" \ + --client-port 4096 --server-port 4096 \ + --client-master-core 8 --server-master-core 8 \ + --client-rx-core 18 --client-tx-core 17 \ + --server-rx-core 19 --server-tx-core 16 \ + --client-worker-core 18 --server-worker-core 19 \ + --message-size 8000000 --buffer-size 8000000 --num-bufs 128 \ + --rx-num-bufs 512 --tx-num-bufs 128 \ + --rx-depth 512 --tx-depth 128 --memory-kind host_pinned \ + --role both --output-dir generated/roce +``` + +`examples/run_spark_bench.sh` uses these profiles directly for its namespace +and benchmark matrix. It does not maintain or mutate Spark-specific base YAMLs. + +## Render any DAQIRI configuration + +The profiles cover common deployment pairs. For HDS, reorder, dynamic-flow, or +application-specific documents, `render` is the general path: it accepts either +a complete document or a bare `daqiri.cfg` mapping and emits the canonical +deterministic serialization. Existing values can be replaced with repeatable +JSON Pointer assignments; assignment values are parsed as YAML 1.2 scalars or +collections. + +Set `TX_PCI`, `RX_PCI`, and `RX_MAC` as described in the raw-Ethernet example +above. Set `TX_IP` and `RX_IP` to the packet-header addresses for your flow: + +```bash +mkdir -p generated +python3 scripts/gen_daqiri_config.py render \ + examples/daqiri_bench_raw_tx_rx.yaml \ + --set /daqiri/cfg/master_core=3 \ + --set /daqiri/cfg/interfaces/0/address="$TX_PCI" \ + --set /daqiri/cfg/interfaces/1/address="$RX_PCI" \ + --set /daqiri/cfg/interfaces/0/tx/queues/0/cpu_core=4 \ + --set /daqiri/cfg/interfaces/1/rx/queues/0/cpu_core=5 \ + --set /bench_tx/0/cpu_core=6 \ + --set /bench_rx/0/cpu_core=7 \ + --set /bench_tx/0/eth_dst_addr="$RX_MAC" \ + --set /bench_tx/0/ip_src_addr="$TX_IP" \ + --set /bench_tx/0/ip_dst_addr="$RX_IP" \ + --output generated/raw.yaml +``` + +An override may replace only an existing path. A misspelled path fails instead +of silently adding a new key. The final document must contain concrete values; +unresolved angle-bracket placeholders are rejected during rendering. + +## Optional hardware-free validation + +Applications parse and validate generated configurations when they call +`daqiri_init()`, so users normally do not need a separate validation step after +generation. Use `daqiri_config_validate` when you want to preflight one or more +files in CI, in a batch, or on a development machine without allocating packet +memory or accessing a NIC. It uses the same C++ parser and common semantic checks +as application initialization: + +```bash +daqiri_config_validate config.yaml another-config.yaml +``` + +The generator checks its own command and profile arguments, but does not +reimplement the complete DAQIRI configuration language. Unknown keys, required +fields, types, ranges, and common semantic constraints remain owned by the C++ +parser and validator. + +### Maintainer checks + +The standard local pull-request check exercises the portable generator tests, +validates retained and generated configurations supported by the validator's +compiled engines, and builds the documentation: + +```bash +scripts/check_pr.sh +``` + +Container release builds validate the same configuration sets with both DPDK +and ibverbs enabled. + +Both configuration check scripts query the validator's compiled engines before +selecting their default cases. Files passed explicitly to +`scripts/check_daqiri_configs.py` are always checked, including files that require +an unavailable engine. + +## Spark verification checklist + +After copying this branch to a Spark and rebuilding the container, first run the +standard pull-request check: + +```bash +scripts/check_pr.sh +``` + +Then run one generated cell per transport. Bring the namespace wire loopback up +for RoCE/TCP/UDP and down for DPDK as described by each harness: + +```bash +# Default namespace, physical p0-to-p1 cable +ETH_DST_ADDR="$(cat /sys/class/net/enP2p1s0f1np1/address)" \ + RUN_SECONDS=10 examples/run_spark_bench.sh dpdk smoke + +# dq_wire_client / dq_wire_server namespaces +RUN_SECONDS=10 examples/run_spark_bench.sh rdma smoke +RUN_SECONDS=10 PAIRS_OVERRIDE=1 examples/run_spark_bench.sh socket-udp smoke +RUN_SECONDS=10 PAIRS_OVERRIDE=1 examples/run_spark_bench.sh socket-tcp smoke +``` + +Finally, exercise every raw multi-queue topology at one payload: + +```bash +ETH_DST_ADDR="$(cat /sys/class/net/enP2p1s0f1np1/address)" \ + PAYLOADS=8000 RUN_SECONDS=10 examples/run_spark_mq_bench.sh +``` + +For each hardware run, retain both application completion output and physical +NIC counter deltas. Non-zero application counts without matching physical +counters can indicate a local shortcut rather than the intended cable path. diff --git a/docs/tutorials/bare-metal-cmake-build.md b/docs/tutorials/bare-metal-cmake-build.md index 607e365a..0d5c52c2 100644 --- a/docs/tutorials/bare-metal-cmake-build.md +++ b/docs/tutorials/bare-metal-cmake-build.md @@ -405,7 +405,7 @@ A successful run prints a stream of `[INFO]` lines followed by an RX/TX rate sum !!! tip "DGX Spark" - On DGX Spark, use the prefilled `daqiri_bench_raw_tx_rx_spark.yaml` instead. Only `eth_dst_addr` needs an edit. See the [DGX Spark profile callout](../benchmarks/raw_benchmarking.md#update-the-loopback-configuration) for the exact MAC-lookup command. + On DGX Spark, use [`daqiri_bench_raw_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx.yaml) to understand the complete TX/RX configuration. Adapt the NIC addresses, use host-pinned memory, select the Spark cores, and set the destination MAC to the RX port. [Generate a raw-Ethernet pair](../config-generation.md#generate-a-raw-ethernet-pair) can apply those parameters for you. !!! note "No NIC available?" @@ -492,9 +492,9 @@ The build recipe above is the same on every supported host. The notes below cove - The integrated **ConnectX-7** appears in `ibv_devinfo` as one or two `mlx5_*` HCAs depending on link configuration. No separate driver install beyond the [DOCA repository setup](#step-1-configure-the-doca-apt-repository) is needed. - GB10 is **compute capability 12.1** (`sm_121`). DAQIRI's default arch list adds `121` automatically when configuring with **CUDA Toolkit 13.0 or newer**; on those toolkits no override is needed. On older toolkits, GB10 is not supported. - - DGX Spark uses **NVLink-C2C unified memory** and has no separate GPU BAR1, so data buffers in YAML configs use `kind: host_pinned` rather than `kind: device`. The DGX-Spark-prefilled YAMLs in `examples/*_spark.yaml` already encode this. + - DGX Spark uses **NVLink-C2C unified memory** and has no separate GPU BAR1, so generate data buffers with `--memory-kind host_pinned` rather than `device`. - `nvidia-peermem` is not used; GPUDirect goes through the dma-buf path enabled by the DPDK patches in [Step 3](#step-3-build-dpdk-with-daqiri-patches). - - For a runnable end-to-end test after the build completes, follow the [DGX Spark profile callout](../benchmarks/raw_benchmarking.md#update-the-loopback-configuration) in Raw Ethernet Benchmarking: the prefilled `daqiri_bench_raw_tx_rx_spark.yaml` and `daqiri_bench_rdma_tx_rx_spark.yaml` need only an `eth_dst_addr` edit. + - For a runnable end-to-end test after the build completes, start with [`daqiri_bench_raw_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx.yaml) or [`daqiri_bench_rdma_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_rdma_tx_rx.yaml), then follow the [DGX Spark profile callout](../benchmarks/raw_benchmarking.md#update-the-loopback-configuration) to adapt or generate a concrete pair from the discovered system parameters. === "IGX Orin + dGPU" diff --git a/docs/tutorials/configuration-walkthrough.md b/docs/tutorials/configuration-walkthrough.md index 34cb938d..3a039a3e 100644 --- a/docs/tutorials/configuration-walkthrough.md +++ b/docs/tutorials/configuration-walkthrough.md @@ -24,6 +24,11 @@ and hardware setups. For a shorter selection guide, start with the [Benchmarking overview](../benchmarks/index.md). With a stream type in mind, read down the questions below and stop at the first one that matches what you're trying to do. Each section names the YAML, the binary that consumes it, and any platform-specific notes. +Use the checked-in YAML files first to understand the configuration and the +relationship between its fields. The generator can then streamline repeatable +system-specific, multi-queue, transform, and cross-host variants without making +those variants the primary teaching examples. + ??? question "1. I want to measure baseline throughput" Pick the stream type that matches your stack (see the [overview](#choosing-the-appropriate-daqiri-stream-type-for-your-setup) above), then the hardware or transport variant. @@ -31,43 +36,29 @@ For a shorter selection guide, start with the [Benchmarking overview](../benchma - **Generic discrete GPU** (template, replace ``): [`daqiri_bench_raw_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx.yaml). This is the file annotated line-by-line in the [walkthrough below](#annotated-walkthrough). - **Four queue closed-loop TX+RX** (template, replace ``): [`daqiri_bench_raw_tx_rx_4q.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_4q.yaml). Uses one application worker per TX/RX queue, with each `bench_tx` entry sending a different UDP flow. - - **DGX Spark / GB10** (prefilled): [`daqiri_bench_raw_tx_rx_spark.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_spark.yaml). `kind: host_pinned` for the integrated GPU, and cores, PCIe addresses, and IPs are prefilled. See the [Spark profile callout](../benchmarks/raw_benchmarking.md#update-the-loopback-configuration) for run details. - - **DGX Spark multi-queue core-scaling matrix** (prefilled): one base config [`daqiri_bench_raw_tx_rx_spark_mq.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_spark_mq.yaml) (the balanced TX=2/RX=2 superset, cores TX → 16,17, RX → 18,19) from which `examples/run_spark_mq_bench.sh` (via `scripts/gen_spark_mq_config.py`) derives the four `(TX, RX)` cells (1,1), (1,2) (RX scaling), (2,1) (TX scaling), and (2,2) (balanced) by pruning queues/flows. All run on `daqiri_bench_raw_gpudirect` at the native 8 KB shape. - - **DGX Spark cross-host** (prefilled, runs on two Sparks): [`daqiri_bench_raw_tx_spark_xhost.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_spark_xhost.yaml) on the TX host and [`daqiri_bench_raw_rx_spark_xhost.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_rx_spark_xhost.yaml) on the RX host. Each host runs `daqiri_bench_raw_gpudirect` against its own half, and cables connect p0↔p0 between the two boxes. Apply the [cross-host network setup](../tutorials/system_configuration.md#cross-host-variant-two-sparks) before running. See the [Cross-host two-DGX-Spark loopback](../benchmarks/raw_benchmarking.md#cross-host-two-dgx-spark-loopback) section for run details. + - **DGX Spark / GB10**: use the generic single-queue example above as the guide, replacing its NIC addresses and cores, using `kind: host_pinned`, and setting `eth_dst_addr` to the RX port's MAC. [`gen_daqiri_config.py`](../config-generation.md#generate-a-raw-ethernet-pair) can apply those Spark parameters; `examples/run_spark_bench.sh` does this for every benchmark cell. + - **DGX Spark multi-queue core-scaling matrix**: use the four-queue example above to understand the queue, memory-region, flow, and application-worker relationships. `examples/run_spark_mq_bench.sh` then generates the 1×1, 1×2, 2×1, and 2×2 cells from the Spark topology. + - **DGX Spark cross-host**: the generic raw example shows the complete TX/RX shape; a cross-host deployment splits it so each host owns one interface and role. Generate independent files with `raw-pair --role both`, copy each file to its host, and start RX before TX. See [configuration generation](../config-generation.md#generate-a-raw-ethernet-pair) and the [cross-host network setup](../tutorials/system_configuration.md#cross-host-variant-two-sparks). - **ConnectX NIC, no cable** (template, replace ``): [`daqiri_bench_raw_hw_loopback_ibverbs.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_hw_loopback_ibverbs.yaml). Uses `engine: "ibverbs"` and `loopback: "hw"` to return TX through the same port's hardware RX steering path. The NIC must remain enumerated without a link and `eth_dst_addr` must be that port's own MAC. - **Runtime named endpoints with payload-only TX buffers** (template, replace ``): [`daqiri_example_named_endpoints_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_example_named_endpoints_tx_rx.yaml) (runs on `daqiri_example_named_endpoints`). Uses the raw ibverbs engine, creates the Ethernet/IPv4/UDP endpoint at runtime, and selects the TX queue at submission. - **No physical NIC available**: [`daqiri_bench_raw_sw_loopback.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_sw_loopback.yaml). `loopback: "sw"`, no NIC required. Useful for first-time build verification, not representative of production performance. - **Raw Ethernet hardware tunnel transforms** run on `daqiri_bench_raw_gpudirect`. Replace placeholders and use a raw DPDK or raw ibverbs build. - - - **VXLAN encap + decap**: [`daqiri_bench_raw_tx_rx_vxlan.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_vxlan.yaml). - - **VLAN push + pop**: [`daqiri_bench_raw_tx_rx_vlan.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_vlan.yaml). - - **GRE encap + decap**: [`daqiri_bench_raw_tx_rx_gre.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_gre.yaml). - - **NVGRE encap + decap**: [`daqiri_bench_raw_tx_rx_nvgre.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_nvgre.yaml). + **Raw Ethernet hardware tunnel transforms** run on `daqiri_bench_raw_gpudirect`. Start with the generic raw example above to understand the queues and flow steering, then use the raw profile with `--transform vlan`, `vxlan`, `gre`, or `nvgre` to generate the complete encap/decap pair; see [configuration generation](../config-generation.md#generate-a-raw-ethernet-pair). To watch the same raw loopback benchmark with live Prometheus and Grafana counters, use the Grafana compose stack described in [Watch live OpenTelemetry metrics in Grafana](../benchmarks/raw_benchmarking.md#watch-live-opentelemetry-metrics-in-grafana). - **Socket (RoCE / RDMA)** (`stream_type: "socket"`, `roce://` endpoints) runs on `daqiri_bench_rdma` (use `--mode {tx,rx,both}`). Configs use `kind: host_pinned` regardless of platform. + **Socket (RoCE / RDMA)** (`stream_type: "socket"`, `roce://` endpoints) runs on `daqiri_bench_rdma` (use `--mode {server,client,both}`). Configs use `kind: host_pinned` regardless of platform. - **Generic** (template, replace IPs): [`daqiri_bench_rdma_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_rdma_tx_rx.yaml). - - **DGX Spark** (prefilled): [`daqiri_bench_rdma_tx_rx_spark.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_rdma_tx_rx_spark.yaml). See [Socket and RDMA Benchmarking](../benchmarks/socket_benchmarking.md#run-the-rdma-roce-benchmark) for namespace and wire-counter run details. - - **DGX Spark netns wire loopback** (prefilled, combined base): [`daqiri_bench_rdma_tx_rx_spark_netns.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_rdma_tx_rx_spark_netns.yaml). Carries both roles. `examples/run_spark_bench.sh` (via `scripts/gen_spark_netns_config.py`) splits it per role and runs each in its own network namespace (`--mode server` / `--mode client`) so RDMA-CM resolves over the wire. See [Socket and RDMA Benchmarking](../benchmarks/socket_benchmarking.md#run-the-rdma-roce-benchmark). - - **DGX Spark cross-host** (prefilled, runs on two Sparks): [`daqiri_bench_rdma_tx_rx_spark_xhost.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_rdma_tx_rx_spark_xhost.yaml). Run with `--mode server` on the RX host and `--mode client` on the TX host. Apply the [cross-host network setup](../tutorials/system_configuration.md#cross-host-variant-two-sparks) before running. See the [Cross-host two-DGX-Spark loopback](../benchmarks/raw_benchmarking.md#cross-host-two-dgx-spark-loopback) section for run details. + - **DGX Spark netns or cross-host**: use the generic RoCE example above to understand the client/server configuration, then adapt the endpoint addresses, cores, host-pinned memory, and per-host roles. Generate the concrete files with `socket-pair --transport roce --role both`; `examples/run_spark_bench.sh` uses this path for its namespace sweep. See [configuration generation](../config-generation.md#generate-udp-tcp-or-roce-roles) and [Socket and RDMA Benchmarking](../benchmarks/socket_benchmarking.md#run-the-rdma-roce-benchmark). **Socket (UDP / TCP)** (`stream_type: "socket"` with `udp://` or `tcp://` endpoints) runs on `daqiri_bench_socket`. The shipped smoke-test configs bind to `127.0.0.1`. See [Socket and RDMA Benchmarking](../benchmarks/socket_benchmarking.md) for namespace-based wire tests. - **UDP**: [`daqiri_bench_socket_udp_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_socket_udp_tx_rx.yaml). - - **UDP, DGX Spark netns wire loopback** (combined base): [`daqiri_bench_socket_udp_tx_rx_spark_netns.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_socket_udp_tx_rx_spark_netns.yaml). Carries both roles. `examples/run_spark_bench.sh` (via `scripts/gen_spark_netns_config.py`) splits it per role and runs each in its own network namespace (`--mode server` / `--mode client`) so same-host IPs cross the wire instead of looping through `lo`. See [Socket and RDMA Benchmarking](../benchmarks/socket_benchmarking.md#run-the-linux-socket-benchmark). - **TCP**: [`daqiri_bench_socket_tcp_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_socket_tcp_tx_rx.yaml). - - **TCP, DGX Spark netns wire loopback** (combined base): [`daqiri_bench_socket_tcp_tx_rx_spark_netns.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_socket_tcp_tx_rx_spark_netns.yaml). Carries both roles. `examples/run_spark_bench.sh` (via `scripts/gen_spark_netns_config.py`) splits it per role and runs each in its own network namespace (`--mode server` / `--mode client`). See [Socket and RDMA Benchmarking](../benchmarks/socket_benchmarking.md#run-the-linux-socket-benchmark). - - **UDP, DGX Spark cross-host** (template, runs on two Sparks): [`daqiri_bench_socket_udp_server_spark_xhost.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_socket_udp_server_spark_xhost.yaml) on the server host and [`daqiri_bench_socket_udp_client_spark_xhost.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_socket_udp_client_spark_xhost.yaml) on the client host. **One file per role** — the socket engine binds every interface in the file at init, so a combined config fails on the host that does not own the other address. Pace the client with `--target-gbps`. Apply the [cross-host network setup](../tutorials/system_configuration.md#cross-host-variant-two-sparks) before running. These produce the UDP rows in [Performance: DGX Spark](../benchmarks/performance-dgx-spark.md#udp-receive-throughput). - - **TCP, DGX Spark cross-host** (template, runs on two Sparks): [`daqiri_bench_socket_tcp_server_spark_xhost.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_socket_tcp_server_spark_xhost.yaml) on the server host and [`daqiri_bench_socket_tcp_client_spark_xhost.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_socket_tcp_client_spark_xhost.yaml) on the client host. **One file per role**, same reason as UDP. TCP self-paces, so no `--target-gbps`. Apply the [cross-host network setup](../tutorials/system_configuration.md#cross-host-variant-two-sparks) before running. These reproduce the retained single-pair TCP measurements in [Performance: DGX Spark](../benchmarks/performance-dgx-spark.md#tcp-receive-throughput). - - Replace ``, ``, ``, and each role-specific - `<*-core>` placeholder in both files before running. Keep the endpoint - addresses and ports consistent across the client/server pair. + - **DGX Spark netns or cross-host**: use the corresponding UDP or TCP example above to understand the socket configuration, then adapt the endpoint addresses, cores, and per-host roles. Generate one concrete file per role with `socket-pair --transport udp` or `--transport tcp`. The socket engine binds every interface in a file, so each host runs only its generated role. See [configuration generation](../config-generation.md#generate-udp-tcp-or-roce-roles). ??? question "I want to measure direct-polling packet latency" Use [`daqiri_bench_raw_latency_ibverbs.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_latency_ibverbs.yaml) with `daqiri_bench_raw_latency`. It sweeps one outstanding packet at a time from 64 through 8192 L2 bytes, using powers of two, with both queues in caller-driven `poll_mode: direct`. diff --git a/docs/tutorials/system_configuration.md b/docs/tutorials/system_configuration.md index c435cbee..faa2df0c 100644 --- a/docs/tutorials/system_configuration.md +++ b/docs/tutorials/system_configuration.md @@ -1574,7 +1574,7 @@ DAQIRI requires an [**NVIDIA SmartNIC**](https://www.nvidia.com/en-us/networking initialization fails if the requested hugepages are unavailable. `kind: "device"` does **not** work on GB10. - See the ready-to-run [`examples/daqiri_bench_raw_tx_rx_spark.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx_spark.yaml) for the complete config. + Use [`daqiri_bench_raw_tx_rx.yaml`](https://github.com/nvidia/daqiri/blob/main/examples/daqiri_bench_raw_tx_rx.yaml) to see how these settings fit into a complete configuration. For repeatable runs, [`scripts/gen_daqiri_config.py`](../config-generation.md#generate-a-raw-ethernet-pair) can apply the Spark addresses, cores, host-pinned memory, and RX MAC for you. ??? info "Why peermem and DMA-BUF don't apply on GB10" @@ -1659,7 +1659,7 @@ DAQIRI requires an [**NVIDIA SmartNIC**](https://www.nvidia.com/en-us/networking ### Step 4: Enable Huge pages (grub drop-in pattern) - Spark composes its `GRUB_CMDLINE_LINUX` from drop-ins under `/etc/default/grub.d/`. Edit a new file rather than `/etc/default/grub` directly so Spark platform updates don't fight your changes. The shipped `daqiri_bench_raw_tx_rx_spark.yaml` needs ~4 GiB of hugepages (kind: HUGE dummy queues + DPDK per-pool overhead); 4 × 1 GiB pages is enough: + Spark composes its `GRUB_CMDLINE_LINUX` from drop-ins under `/etc/default/grub.d/`. Edit a new file rather than `/etc/default/grub` directly so Spark platform updates don't fight your changes. The generated Spark DPDK profile needs ~4 GiB of hugepages (kind: HUGE dummy queues + DPDK per-pool overhead); 4 × 1 GiB pages is enough: ```bash cat << 'EOF' | sudo tee /etc/default/grub.d/daqiri-tuning.cfg diff --git a/examples/daqiri_bench_raw_rx_spark_xhost.yaml b/examples/daqiri_bench_raw_rx_spark_xhost.yaml deleted file mode 100644 index d86f213b..00000000 --- a/examples/daqiri_bench_raw_rx_spark_xhost.yaml +++ /dev/null @@ -1,58 +0,0 @@ -# DGX Spark (GB10) cross-host RX-side config for daqiri_bench_raw_gpudirect. -# Companion to daqiri_bench_raw_tx_spark_xhost.yaml on the peer TX host. Both -# hosts must be configured per the DGX Spark profile in -# docs/tutorials/system_configuration.md, with the chosen p0 port cross-cabled -# host-to-host (no chassis QSFP loop). -# -# Spark substitutions baked in here: -# - rx_port BDF = 0000:01:00.0 (p0); change if your p0 sits elsewhere -# - kind: host_pinned (GB10 dma-buf path; nvidia-peermem N/A on Spark) -# - master_core: 8; cpu_core: 18 for the DAQIRI RX queue and 19 for the app -# RX worker -- separate cores so they don't contend (isolated big-cluster -# X925 16-19) -# - flow match: udp_src/dst = 4096 -- same UDP tuple the TX side sends to -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "raw" - engine: "ibverbs" - master_core: 8 - debug: false - log_level: "info" - loopback: "" - - memory_regions: - - name: "Data_RX_GPU" - kind: "host_pinned" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - interfaces: - - name: "rx_port" - address: 0000:01:00.0 - rx: - flow_isolation: true - queues: - - name: "rq_q_0" - id: 0 - cpu_core: 18 - batch_size: 10240 - memory_regions: - - "Data_RX_GPU" - flows: - - name: "flow_0" - id: 1 - action: - type: queue - id: 0 - match: - udp_src: 4096 - udp_dst: 4096 - -bench_rx: - interface_name: "rx_port" - cpu_core: 19 diff --git a/examples/daqiri_bench_raw_tx_rx_gre.yaml b/examples/daqiri_bench_raw_tx_rx_gre.yaml deleted file mode 100644 index bbe15d50..00000000 --- a/examples/daqiri_bench_raw_tx_rx_gre.yaml +++ /dev/null @@ -1,92 +0,0 @@ -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "raw" - master_core: 3 - debug: false - log_level: "info" - loopback: "" - - memory_regions: - - name: "Data_TX_GPU" - kind: "device" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - name: "Data_RX_GPU" - kind: "device" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - interfaces: - - name: "tx_port" - address: <0000:00:00.0> - tx: - queues: - - name: "tx_q_0" - id: 0 - batch_size: 10240 - cpu_core: 11 - memory_regions: - - "Data_TX_GPU" - offloads: - - "tx_eth_src" - flows: - - name: "gre_encap" - id: 102 - actions: - - type: tunnel_encap - tunnel: - type: gre - outer_eth_src: <00:00:00:00:00:00> - outer_eth_dst: <00:00:00:00:00:00> - outer_ipv4_src: <192.0.2.1> - outer_ipv4_dst: <192.0.2.2> - gre_protocol: 0x0800 - match: - udp_src: 4096 - udp_dst: 4096 - - name: "rx_port" - address: <0000:00:00.0> - rx: - flow_isolation: true - queues: - - name: "rq_q_0" - id: 0 - cpu_core: 9 - batch_size: 10240 - memory_regions: - - "Data_RX_GPU" - flows: - - name: "gre_decap" - id: 102 - actions: - - type: tunnel_decap - tunnel: - type: gre - outer_eth_src: <00:00:00:00:00:00> - outer_eth_dst: <00:00:00:00:00:00> - outer_ipv4_src: <192.0.2.1> - outer_ipv4_dst: <192.0.2.2> - gre_protocol: 0x0800 - - type: queue - id: 0 - -bench_rx: -- interface_name: "rx_port" - cpu_core: 8 - -bench_tx: -- interface_name: "tx_port" - cpu_core: 10 - batch_size: 10240 - payload_size: 8000 - header_size: 64 - eth_dst_addr: <00:00:00:00:00:00> - ip_src_addr: <1.2.3.4> - ip_dst_addr: <5.6.7.8> - udp_src_port: 4096 - udp_dst_port: 4096 diff --git a/examples/daqiri_bench_raw_tx_rx_nvgre.yaml b/examples/daqiri_bench_raw_tx_rx_nvgre.yaml deleted file mode 100644 index cc9b7e72..00000000 --- a/examples/daqiri_bench_raw_tx_rx_nvgre.yaml +++ /dev/null @@ -1,94 +0,0 @@ -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "raw" - master_core: 3 - debug: false - log_level: "info" - loopback: "" - - memory_regions: - - name: "Data_TX_GPU" - kind: "device" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - name: "Data_RX_GPU" - kind: "device" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - interfaces: - - name: "tx_port" - address: <0000:00:00.0> - tx: - queues: - - name: "tx_q_0" - id: 0 - batch_size: 10240 - cpu_core: 11 - memory_regions: - - "Data_TX_GPU" - offloads: - - "tx_eth_src" - flows: - - name: "nvgre_encap" - id: 103 - actions: - - type: tunnel_encap - tunnel: - type: nvgre - outer_eth_src: <00:00:00:00:00:00> - outer_eth_dst: <00:00:00:00:00:00> - outer_ipv4_src: <192.0.2.1> - outer_ipv4_dst: <192.0.2.2> - tni: 100 - flow_id: 0 - match: - udp_src: 4096 - udp_dst: 4096 - - name: "rx_port" - address: <0000:00:00.0> - rx: - flow_isolation: true - queues: - - name: "rq_q_0" - id: 0 - cpu_core: 9 - batch_size: 10240 - memory_regions: - - "Data_RX_GPU" - flows: - - name: "nvgre_decap" - id: 103 - actions: - - type: tunnel_decap - tunnel: - type: nvgre - outer_eth_src: <00:00:00:00:00:00> - outer_eth_dst: <00:00:00:00:00:00> - outer_ipv4_src: <192.0.2.1> - outer_ipv4_dst: <192.0.2.2> - tni: 100 - flow_id: 0 - - type: queue - id: 0 - -bench_rx: -- interface_name: "rx_port" - cpu_core: 8 - -bench_tx: -- interface_name: "tx_port" - cpu_core: 10 - batch_size: 10240 - payload_size: 8000 - header_size: 64 - eth_dst_addr: <00:00:00:00:00:00> - ip_src_addr: <1.2.3.4> - ip_dst_addr: <5.6.7.8> - udp_src_port: 4096 - udp_dst_port: 4096 diff --git a/examples/daqiri_bench_raw_tx_rx_spark.yaml b/examples/daqiri_bench_raw_tx_rx_spark.yaml deleted file mode 100644 index 42791d83..00000000 --- a/examples/daqiri_bench_raw_tx_rx_spark.yaml +++ /dev/null @@ -1,103 +0,0 @@ -# DGX Spark (GB10) ready-to-run config for daqiri_bench_raw_gpudirect. -# Templated version (with placeholders) is in -# daqiri_bench_raw_tx_rx.yaml. -# -# Spark substitutions baked in here: -# - PCIe addresses: this file is configured for an over-the-wire loopback -- -# tx_port (0000:01:00.0) and rx_port (0002:01:00.1) are different physical -# ports (p0 / p1) cross-cabled by the chassis QSFP loop, so traffic crosses -# the cable. To verify the port topology or set up a different loopback -# (e.g. on-chip), see the "Port topology" section of the "DGX Spark" tab in -# docs/tutorials/system_configuration.md. -# - kind: host_pinned, NOT device. GB10 has unified memory; nvidia_peermem -# does not load and CUDA reports DMA_BUF_SUPPORTED=0. PR #41 made -# host_pinned the working GPUDirect path on Spark (~94 Gbps unicast). -# - cpu_core: DAQIRI queue threads use 17 (TX) / 18 (RX); the benchmark app -# workers use the remaining big cores 16 (TX) / 19 (RX) so the app and -# DAQIRI poll threads never share a core. All four are big-cluster X925 -# cores 16-19 isolated by the daqiri-tuning grub drop-in. -# - master_core: 8 (a non-isolated big core; any 0-15 works). -# - bench_tx ip_src/ip_dst: match the daqiri-tx / daqiri-rx nmcli profiles -# (1.1.1.1/24 and 2.2.2.2/24, MTU 9000). -# - eth_dst_addr: your rx_port's own MAC; fill per-system, e.g.: -# cat /sys/class/net/enP2p1s0f1np1/address -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "raw" - # This report configuration exercises the DPDK raw-Ethernet engine. Raw - # otherwise defaults to ibverbs when that engine is compiled. - engine: "dpdk" - master_core: 8 - debug: false - log_level: "info" - loopback: "" - - memory_regions: - - name: "Data_TX_GPU" - kind: "host_pinned" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - name: "Data_RX_GPU" - kind: "host_pinned" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - interfaces: - - name: "tx_port" - address: 0000:01:00.0 - tx: - queues: - - name: "tx_q_0" - id: 0 - batch_size: 10240 - cpu_core: 17 - memory_regions: - - "Data_TX_GPU" - offloads: - - "tx_eth_src" - - name: "rx_port" - # p1 -- a DIFFERENT physical port than tx_port (p0), cross-cabled to it, so - # traffic crosses the wire. The same physical port (e.g. 0002:01:00.0, also - # p0) would loop on-chip instead. Verify your p0/p1 mapping (see header). - address: 0002:01:00.1 - rx: - flow_isolation: true - queues: - - name: "rq_q_0" - id: 0 - cpu_core: 18 - batch_size: 10240 - memory_regions: - - "Data_RX_GPU" - flows: - - name: "flow_0" - id: 0 - action: - type: queue - id: 0 - match: - udp_src: 4096 - udp_dst: 4096 - -bench_rx: - interface_name: "rx_port" - cpu_core: 19 - -bench_tx: - interface_name: "tx_port" - cpu_core: 16 - batch_size: 10240 - payload_size: 8000 - header_size: 64 - # rx_port's (p1) own MAC; fill per-system, e.g. cat /sys/class/net/enP2p1s0f1np1/address - eth_dst_addr: <00:00:00:00:00:00> - ip_src_addr: 1.1.1.1 - ip_dst_addr: 2.2.2.2 - udp_src_port: 4096 - udp_dst_port: 4096 diff --git a/examples/daqiri_bench_raw_tx_rx_spark_mq.yaml b/examples/daqiri_bench_raw_tx_rx_spark_mq.yaml deleted file mode 100644 index 2afdd1c2..00000000 --- a/examples/daqiri_bench_raw_tx_rx_spark_mq.yaml +++ /dev/null @@ -1,168 +0,0 @@ -# DGX Spark (GB10) multi-queue core-scaling base -- the (TX=2, RX=2) superset. -# -# This is the single checked-in config for the multi-queue scaling sweep. The -# matrix cells (TX,RX) = (1,1),(1,2),(2,1),(2,2) are derived from this file at -# sweep time by scripts/gen_spark_mq_config.py (invoked by run_spark_mq_bench.sh), -# which prunes queues/flows/memory-regions/bench entries down to each cell. To -# run a single cell by hand, e.g. (1,2): -# scripts/gen_spark_mq_config.py examples/daqiri_bench_raw_tx_rx_spark_mq.yaml \ -# --tx 1 --rx 2 --payload 8000 --eth-dst > cell.yaml -# daqiri_bench_raw_gpudirect cell.yaml --seconds 30 -# As shown below (no pruning) it is the balanced (2,2) cell. -# -# Shared Spark substitutions (same as daqiri_bench_raw_tx_rx_spark.yaml): -# - over-the-wire loopback: tx_port 0000:01:00.0 (p0) -> rx_port 0002:01:00.1 -# (p1) are different physical ports cross-cabled by the chassis QSFP loop, so -# traffic crosses the cable. See the "Port topology" section of the "DGX -# Spark" tab in docs/tutorials/system_configuration.md. -# - kind: host_pinned (GB10 unified memory; nvidia_peermem does not load). -# - Each queue's poller and worker run on SEPARATE isolated X925 cores, kept in -# the same hardware cluster where possible so the poller->worker ring handoff -# stays in-cluster (a cross-cluster handoff roughly halves the small-payload -# op-rate). The two performance clusters are cpu 15-19 and cpu 5-9. -# NOTE: queue POLLERS are DPDK EAL lcores and must avoid any core the CUDA -# runtime pins helper threads to -- CUDA lands on the low cores of the 5-9 -# cluster (seen on 5 and 6, unpredictably) and an EAL lcore on a CUDA-occupied -# core never launches, hanging the bench. So pollers use only 15-19 + core 9 -# (never 5/6/8). Bench WORKERS are plain pinned pthreads and coexist with CUDA -# fine, so they may use 5/6. -# TX q0: poller 16 / worker 15 (cluster 15-19, local) -# RX q0: poller 18 / worker 17 (cluster 15-19, local) -- 1t1r lives here, -# matching the single-queue config -# RX q1: poller 9 / worker 7 (cluster 5-9, local; RX kept local) -# TX q1: poller 19 / worker 6 (poller in 15-19, worker in 5-9: this one -# handoff crosses clusters, but TX is not -# the small-payload bottleneck so it is ok) -# This needs the EXPANDED isolcpus=5-9,15-19 (all 10 big cores). -# - master_core 8 (big core). -# - native shape: 8 KB payload (payload_size 8000 + header_size 64). -# - eth_dst_addr: your rx_port's (p1) own MAC; fill per-system, e.g. -# cat /sys/class/net/enP2p1s0f1np1/address -# -# Cell (2,2): two TX queues (pollers 16,19) feed two RX queues (pollers 18,9), one -# flow per pair -- tx_q_0 on (4096,4096) -> rx_q_0 and tx_q_1 on (4097,4097) -> -# rx_q_1. Fully balanced; the top of the scaling matrix and should beat (1,1) -# when a single core is the bottleneck. -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "raw" - # This report configuration exercises the DPDK raw-Ethernet engine. Raw - # otherwise defaults to ibverbs when that engine is compiled. - engine: "dpdk" - master_core: 8 - debug: false - log_level: "info" - loopback: "" - - memory_regions: - - name: "Data_TX_GPU_0" - kind: "host_pinned" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - name: "Data_TX_GPU_1" - kind: "host_pinned" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - name: "Data_RX_GPU_0" - kind: "host_pinned" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - name: "Data_RX_GPU_1" - kind: "host_pinned" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - interfaces: - - name: "tx_port" - address: "0000:01:00.0" # quoted: PyYAML (gen_spark_mq_config.py) else reads the BDF as base-60 - tx: - queues: - - name: "tx_q_0" - id: 0 - batch_size: 10240 - cpu_core: 16 - memory_regions: - - "Data_TX_GPU_0" - offloads: - - "tx_eth_src" - - name: "tx_q_1" - id: 1 - batch_size: 10240 - cpu_core: 19 - memory_regions: - - "Data_TX_GPU_1" - offloads: - - "tx_eth_src" - - name: "rx_port" - address: "0002:01:00.1" # quoted: PyYAML (gen_spark_mq_config.py) else reads the BDF as base-60 - rx: - flow_isolation: true - queues: - - name: "rx_q_0" - id: 0 - cpu_core: 18 - batch_size: 10240 - memory_regions: - - "Data_RX_GPU_0" - - name: "rx_q_1" - id: 1 - cpu_core: 9 - batch_size: 10240 - memory_regions: - - "Data_RX_GPU_1" - flows: - - name: "flow_0" - id: 0 - action: - type: queue - id: 0 - match: - udp_src: 4096 - udp_dst: 4096 - - name: "flow_1" - id: 1 - action: - type: queue - id: 1 - match: - udp_src: 4097 - udp_dst: 4097 - -bench_rx: -- interface_name: "rx_port" - queue_id: 0 - cpu_core: 17 # RX q0 worker (cluster A); its queue poller is on 18 -- interface_name: "rx_port" - queue_id: 1 - cpu_core: 7 # RX q1 worker (cluster B); its queue poller is on 9 - -bench_tx: -- interface_name: "tx_port" - queue_id: 0 - cpu_core: 15 # TX q0 worker (cluster A); its queue poller is on 16 - batch_size: 10240 - payload_size: 8000 - header_size: 64 - eth_dst_addr: <00:00:00:00:00:00> - ip_src_addr: 1.1.1.1 - ip_dst_addr: 2.2.2.2 - udp_src_port: 4096 - udp_dst_port: 4096 -- interface_name: "tx_port" - queue_id: 1 - cpu_core: 6 # TX q1 worker (cluster 5-9); its poller is on 19 (cross-cluster) - batch_size: 10240 - payload_size: 8000 - header_size: 64 - eth_dst_addr: <00:00:00:00:00:00> - ip_src_addr: 1.1.1.1 - ip_dst_addr: 2.2.2.2 - udp_src_port: 4097 - udp_dst_port: 4097 diff --git a/examples/daqiri_bench_raw_tx_rx_vlan.yaml b/examples/daqiri_bench_raw_tx_rx_vlan.yaml deleted file mode 100644 index 7f485df0..00000000 --- a/examples/daqiri_bench_raw_tx_rx_vlan.yaml +++ /dev/null @@ -1,82 +0,0 @@ -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "raw" - master_core: 3 - debug: false - log_level: "info" - loopback: "" - - memory_regions: - - name: "Data_TX_GPU" - kind: "device" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - name: "Data_RX_GPU" - kind: "device" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - interfaces: - - name: "tx_port" - address: <0000:00:00.0> - tx: - queues: - - name: "tx_q_0" - id: 0 - batch_size: 10240 - cpu_core: 11 - memory_regions: - - "Data_TX_GPU" - offloads: - - "tx_eth_src" - flows: - - name: "vlan_push" - id: 101 - actions: - - type: vlan_push - vlan_id: 100 - pcp: 0 - dei: 0 - ethertype: 0x8100 - match: - udp_src: 4096 - udp_dst: 4096 - - name: "rx_port" - address: <0000:00:00.0> - rx: - flow_isolation: true - queues: - - name: "rq_q_0" - id: 0 - cpu_core: 9 - batch_size: 10240 - memory_regions: - - "Data_RX_GPU" - flows: - - name: "vlan_pop" - id: 101 - actions: - - type: vlan_pop - - type: queue - id: 0 - -bench_rx: -- interface_name: "rx_port" - cpu_core: 8 - -bench_tx: -- interface_name: "tx_port" - cpu_core: 10 - batch_size: 10240 - payload_size: 8000 - header_size: 64 - eth_dst_addr: <00:00:00:00:00:00> - ip_src_addr: <1.2.3.4> - ip_dst_addr: <5.6.7.8> - udp_src_port: 4096 - udp_dst_port: 4096 diff --git a/examples/daqiri_bench_raw_tx_rx_vxlan.yaml b/examples/daqiri_bench_raw_tx_rx_vxlan.yaml deleted file mode 100644 index 87c03352..00000000 --- a/examples/daqiri_bench_raw_tx_rx_vxlan.yaml +++ /dev/null @@ -1,96 +0,0 @@ -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "raw" - master_core: 3 - debug: false - log_level: "info" - loopback: "" - - memory_regions: - - name: "Data_TX_GPU" - kind: "device" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - name: "Data_RX_GPU" - kind: "device" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - interfaces: - - name: "tx_port" - address: <0000:00:00.0> - tx: - queues: - - name: "tx_q_0" - id: 0 - batch_size: 10240 - cpu_core: 11 - memory_regions: - - "Data_TX_GPU" - offloads: - - "tx_eth_src" - flows: - - name: "vxlan_encap" - id: 100 - actions: - - type: tunnel_encap - tunnel: - type: vxlan - outer_eth_src: <00:00:00:00:00:00> - outer_eth_dst: <00:00:00:00:00:00> - outer_ipv4_src: <192.0.2.1> - outer_ipv4_dst: <192.0.2.2> - outer_udp_src: 49152 - outer_udp_dst: 4789 - vni: 100 - match: - udp_src: 4096 - udp_dst: 4096 - - name: "rx_port" - address: <0000:00:00.0> - rx: - flow_isolation: true - queues: - - name: "rq_q_0" - id: 0 - cpu_core: 9 - batch_size: 10240 - memory_regions: - - "Data_RX_GPU" - flows: - - name: "vxlan_decap" - id: 100 - actions: - - type: tunnel_decap - tunnel: - type: vxlan - outer_eth_src: <00:00:00:00:00:00> - outer_eth_dst: <00:00:00:00:00:00> - outer_ipv4_src: <192.0.2.1> - outer_ipv4_dst: <192.0.2.2> - outer_udp_src: 49152 - outer_udp_dst: 4789 - vni: 100 - - type: queue - id: 0 - -bench_rx: -- interface_name: "rx_port" - cpu_core: 8 - -bench_tx: -- interface_name: "tx_port" - cpu_core: 10 - batch_size: 10240 - payload_size: 8000 - header_size: 64 - eth_dst_addr: <00:00:00:00:00:00> - ip_src_addr: <1.2.3.4> - ip_dst_addr: <5.6.7.8> - udp_src_port: 4096 - udp_dst_port: 4096 diff --git a/examples/daqiri_bench_raw_tx_spark_xhost.yaml b/examples/daqiri_bench_raw_tx_spark_xhost.yaml deleted file mode 100644 index 94b43e4b..00000000 --- a/examples/daqiri_bench_raw_tx_spark_xhost.yaml +++ /dev/null @@ -1,60 +0,0 @@ -# DGX Spark (GB10) cross-host TX-side config for daqiri_bench_raw_gpudirect. -# Companion to daqiri_bench_raw_rx_spark_xhost.yaml on the peer RX host. Both -# hosts must be configured per the DGX Spark profile in -# docs/tutorials/system_configuration.md, with the chosen p0 port cross-cabled -# host-to-host (no chassis QSFP loop). -# -# Spark substitutions baked in here: -# - tx_port BDF = 0000:01:00.0 (p0); change if your p0 sits elsewhere -# - kind: host_pinned (GB10 dma-buf path; nvidia-peermem N/A on Spark) -# - master_core: 8; cpu_core: 17 for the DAQIRI TX queue and 16 for the app -# TX worker -- separate cores so they don't contend (isolated big-cluster -# X925 16-19) -# - engine: ibverbs leaves ARP on the kernel path. This TX-only interface has -# no DAQIRI RX queues, so no RX catch-all can divert ARP from Linux. -# eth_dst_addr is omitted, so DAQIRI resolves 2.2.2.2 through the kernel -# route and neighbour tables. -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "raw" - engine: "ibverbs" - master_core: 8 - debug: false - log_level: "info" - loopback: "" - - memory_regions: - - name: "Data_TX_GPU" - kind: "host_pinned" - affinity: 0 - num_bufs: 51200 - buf_size: 8064 - - interfaces: - - name: "tx_port" - address: 0000:01:00.0 - tx: - queues: - - name: "tx_q_0" - id: 0 - batch_size: 4096 - cpu_core: 17 - memory_regions: - - "Data_TX_GPU" - offloads: - - "tx_eth_src" - -bench_tx: - interface_name: "tx_port" - cpu_core: 16 - batch_size: 4096 - payload_size: 8000 - header_size: 64 - ip_src_addr: 1.1.1.1 - ip_dst_addr: 2.2.2.2 - udp_src_port: 4096 - udp_dst_port: 4096 diff --git a/examples/daqiri_bench_rdma_tx_rx_spark.yaml b/examples/daqiri_bench_rdma_tx_rx_spark.yaml deleted file mode 100644 index 33ac6f18..00000000 --- a/examples/daqiri_bench_rdma_tx_rx_spark.yaml +++ /dev/null @@ -1,105 +0,0 @@ -# DGX Spark (GB10) ready-to-run config for daqiri_bench_rdma. -# Templated version (different IPs, different cores) is in -# daqiri_bench_rdma_tx_rx.yaml. -# -# Spark substitutions: -# - IPs: 1.1.1.1 (client/TX) and 2.2.2.2 (server/RX), matching the -# daqiri-tx / daqiri-rx nmcli profiles documented in the system -# configuration tutorial DGX Spark tab. -# - cpu_core values for DAQIRI queue threads and benchmark app role threads -# are pulled from the isolated big-cluster X925 cores 16-19 (see grub -# drop-in /etc/default/grub.d/daqiri-tuning.cfg). -# - master_core: 8 (non-isolated big core). -# - kind: host_pinned (already correct upstream and required on GB10 -# where peermem is N/A and DMA-BUF is unreachable). -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "socket" - master_core: 8 - debug: false - log_level: "info" - - memory_regions: - - name: "DATA_RX_GPU_SERVER" - kind: "host_pinned" - affinity: 0 - num_bufs: 20 - buf_size: 9000000 - - name: "DATA_TX_GPU_SERVER" - kind: "host_pinned" - affinity: 0 - num_bufs: 20 - buf_size: 9000000 - - name: "DATA_TX_GPU_CLIENT" - kind: "host_pinned" - affinity: 0 - num_bufs: 20 - buf_size: 90000000 - - name: "DATA_RX_GPU_CLIENT" - kind: "host_pinned" - affinity: 0 - num_bufs: 20 - buf_size: 90000000 - - interfaces: - - name: my_client - address: 1.1.1.1 - socket_config: - mode: client - local_addr: "roce://1.1.1.1" - roce_config: - transport_mode: RC - tx: - queues: - - name: "Client_TX_Queue" - id: 0 - batch_size: 1 - cpu_core: 17 - rx: - queues: - - name: "Client_RX_Queue" - id: 0 - cpu_core: 18 - batch_size: 1 - - name: my_server - address: 2.2.2.2 - socket_config: - mode: server - local_addr: "roce://2.2.2.2:4096" - roce_config: - transport_mode: RC - rx: - queues: - - name: "Server_RX_Queue" - id: 0 - cpu_core: 19 - batch_size: 1 - tx: - queues: - - name: "Server_TX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - -rdma_bench_server: - server_address: 2.2.2.2 - server_port: 4096 - message_size: 8000000 - send: true - receive: true - cpu_core: 19 - server: true - -rdma_bench_client: - message_size: 8000000 - client_address: 1.1.1.1 - server_address: 2.2.2.2 - server_port: 4096 - receive: true - send: true - cpu_core: 17 - server: false diff --git a/examples/daqiri_bench_rdma_tx_rx_spark_netns.yaml b/examples/daqiri_bench_rdma_tx_rx_spark_netns.yaml deleted file mode 100644 index 0a110872..00000000 --- a/examples/daqiri_bench_rdma_tx_rx_spark_netns.yaml +++ /dev/null @@ -1,141 +0,0 @@ -# DGX Spark (GB10) RDMA bench — combined base for the netns wire loopback. -# -# This is the single checked-in config for the RoCE wire-loopback sweep. It -# carries BOTH roles (the my_server / my_client interfaces, their memory regions, -# and the rdma_bench_server / rdma_bench_client sections). A single process can't -# run it as-is across namespaces (it would init the peer namespace's IP it does -# not own), so scripts/gen_spark_netns_config.py splits it to one role at sweep -# time and run_spark_bench.sh runs each role in its own namespace. To split by -# hand: -# scripts/gen_spark_netns_config.py examples/daqiri_bench_rdma_tx_rx_spark_netns.yaml \ -# --role server > server.yaml -# ip netns exec dq_wire_server daqiri_bench_rdma server.yaml --seconds N --mode server -# -# IPs / RDMA devices match setup_spark_wire_loopback_netns.sh defaults: -# SERVER_IP=10.250.0.2 SERVER_NS=dq_wire_server SERVER_RDMA=roceP2p1s0f1 -# CLIENT_IP=10.250.0.1 CLIENT_NS=dq_wire_client CLIENT_RDMA=rocep1s0f0 -# cpu_core values use the isolated X925 big-cluster cores 16-19 (isolcpus). -# -# num_bufs must be >= the flow-control window (rx_depth for RX, tx_depth for TX) -# below, or is_tx_burst_available() caps the effective window at num_bufs and -# throttles small-message op-rate. buf_size is 10 MB: comfortably above the 8 MB -# largest sweep message, while bounding pinned memory (128 * 10 MB * 2 MRs ~= -# 2.5 GB/process) instead of the ~23 GB a 90 MB buffer would need at this depth. -# run_spark_bench.sh's rdma sweep rewrites num_bufs/buf_size/{rx,tx}_depth per -# message size (deep window at small sizes, memory-capped at large ones). -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "socket" - master_core: 8 - debug: false - log_level: "info" - - memory_regions: - - name: "DATA_RX_GPU_SERVER" - kind: "host_pinned" - affinity: 0 - num_bufs: 128 - buf_size: 10000000 - - name: "DATA_TX_GPU_SERVER" - kind: "host_pinned" - affinity: 0 - num_bufs: 128 - buf_size: 10000000 - - name: "DATA_TX_GPU_CLIENT" - kind: "host_pinned" - affinity: 0 - num_bufs: 128 - buf_size: 10000000 - - name: "DATA_RX_GPU_CLIENT" - kind: "host_pinned" - affinity: 0 - num_bufs: 128 - buf_size: 10000000 - - interfaces: - - name: my_server - address: 10.250.0.2 - socket_config: - mode: server - local_addr: "roce://10.250.0.2:4096" - roce_config: - transport_mode: RC - rx: - queues: - - name: "Server_RX_Queue" - id: 0 - cpu_core: 19 - batch_size: 1 - tx: - queues: - - name: "Server_TX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - - name: my_client - address: 10.250.0.1 - socket_config: - mode: client - # Client binds its own RoCE address here; the server peer endpoint lives - # in the rdma_bench_client app config (server_address/server_port). Since - # #156 a RoCE client's socket_config.remote_addr is rejected by daqiri_init. - local_addr: "roce://10.250.0.1" - roce_config: - transport_mode: RC - tx: - queues: - - name: "Client_TX_Queue" - id: 0 - cpu_core: 17 - batch_size: 1 - rx: - queues: - - name: "Client_RX_Queue" - id: 0 - cpu_core: 18 - batch_size: 1 - -rdma_bench_server: - # Bench worker thread (PR #149) on a SEPARATE core from the DAQIRI RoCE mgr - # thread. The mgr thread pins to the TX queue core (Server_TX_Queue = 16) and - # hands off through a ring, so co-locating the worker there livelocks at small - # message sizes (same failure mode as DPDK small-batch). Put the worker on the - # server's other isolated core (the RX queue core, 19). - cpu_core: 19 - server_address: 10.250.0.2 - server_port: 4096 - message_size: 8000000 - # Flow-control windows (PR #144). Pre-post rx_depth receives before sending and - # bound the transmit side by tx_depth so small-message SENDs don't outrun the - # peer's posted receives and trigger RNR NACKs. Keep <= the MR num_bufs above. - rx_depth: 128 - tx_depth: 128 - # One-way (unidirectional) by default, matching the DPDK and socket benches: - # the server only receives, the client only sends. Set send: true here (and - # receive: true on the client) for a bidirectional test. - send: false - receive: true - server: true - -rdma_bench_client: - # Bench worker on a separate core from the mgr thread (Client_TX_Queue = 17); - # worker on the client's RX queue core (18). See the server note above. - cpu_core: 18 - message_size: 8000000 - # Flow-control windows (PR #144). Pre-post rx_depth receives before sending and - # bound the transmit side by tx_depth so small-message SENDs don't outrun the - # peer's posted receives and trigger RNR NACKs. Keep <= the MR num_bufs above. - rx_depth: 128 - tx_depth: 128 - client_address: 10.250.0.1 - server_address: 10.250.0.2 - server_port: 4096 - # One-way (see server note above): the client only sends. Set receive: true for - # a bidirectional test. - receive: false - send: true - server: false diff --git a/examples/daqiri_bench_rdma_tx_rx_spark_xhost.yaml b/examples/daqiri_bench_rdma_tx_rx_spark_xhost.yaml deleted file mode 100644 index ec94f4e2..00000000 --- a/examples/daqiri_bench_rdma_tx_rx_spark_xhost.yaml +++ /dev/null @@ -1,117 +0,0 @@ -# DGX Spark (GB10) cross-host config for daqiri_bench_rdma. -# Adapts the single-host daqiri_bench_rdma_tx_rx_spark.yaml to a two-host -# setup with each side's p0 cross-cabled to the peer's p0. Both hosts must -# be configured per the DGX Spark profile in -# docs/tutorials/system_configuration.md, except the IP assignment is -# split across hosts: put 1.1.1.1/24 on the client host's p0 and 2.2.2.2/24 -# on the server host's p0 (instead of both addresses on one machine). -# -# Before running, apply cross-host L3 on both boxes: -# sudo scripts/setup_spark_xhost_net.sh --role tx --peer-mac # TX host -# sudo scripts/setup_spark_xhost_net.sh --role rx --peer-mac # RX host -# See docs/tutorials/system_configuration.md#cross-host-variant-two-sparks. -# -# Run with --mode client on the TX host and --mode server on the RX host: -# server (RX): sudo ./daqiri_bench_rdma --mode server --seconds 30 -# client (TX): sudo ./daqiri_bench_rdma --mode client --seconds 30 -# -# Regression test for cross-host RDMA receive readiness (#113, #174). -# Spark substitutions baked in here: -# - IPs: 1.1.1.1 (client/TX p0) and 2.2.2.2 (server/RX p0) -# - cpu_core values for DAQIRI queues and benchmark app role threads from -# isolated big-cluster X925 16-19; master_core: 8. Each host runs only one -# role, so the app thread takes a big core its local queues don't use: -# client host queues 17/18 -> client app 16; server host queues 16/19 -> -# server app 18. -# - kind: host_pinned (required upstream on GB10; peermem N/A, dma-buf used) -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "socket" - master_core: 8 - debug: false - log_level: "info" - - memory_regions: - - name: "DATA_RX_GPU_SERVER" - kind: "host_pinned" - affinity: 0 - num_bufs: 20 - buf_size: 9000000 - - name: "DATA_TX_GPU_SERVER" - kind: "host_pinned" - affinity: 0 - num_bufs: 20 - buf_size: 9000000 - - name: "DATA_TX_GPU_CLIENT" - kind: "host_pinned" - affinity: 0 - num_bufs: 20 - buf_size: 90000000 - - name: "DATA_RX_GPU_CLIENT" - kind: "host_pinned" - affinity: 0 - num_bufs: 20 - buf_size: 90000000 - - interfaces: - - name: my_client - address: 1.1.1.1 - socket_config: - mode: client - local_addr: "roce://1.1.1.1" - roce_config: - transport_mode: RC - tx: - queues: - - name: "Client_TX_Queue" - id: 0 - batch_size: 1 - cpu_core: 17 - rx: - queues: - - name: "Client_RX_Queue" - id: 0 - cpu_core: 18 - batch_size: 1 - - name: my_server - address: 2.2.2.2 - socket_config: - mode: server - local_addr: "roce://2.2.2.2:4096" - roce_config: - transport_mode: RC - rx: - queues: - - name: "Server_RX_Queue" - id: 0 - cpu_core: 19 - batch_size: 1 - tx: - queues: - - name: "Server_TX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - -rdma_bench_server: - server_address: 2.2.2.2 - server_port: 4096 - message_size: 8000000 - send: true - receive: true - cpu_core: 18 - server: true - -rdma_bench_client: - message_size: 8000000 - client_address: 1.1.1.1 - server_address: 2.2.2.2 - server_port: 4096 - receive: true - send: true - cpu_core: 16 - server: false diff --git a/examples/daqiri_bench_socket_tcp_client_spark_xhost.yaml b/examples/daqiri_bench_socket_tcp_client_spark_xhost.yaml deleted file mode 100644 index 89689b5a..00000000 --- a/examples/daqiri_bench_socket_tcp_client_spark_xhost.yaml +++ /dev/null @@ -1,81 +0,0 @@ -# DGX Spark (GB10) cross-host TCP client config for daqiri_bench_socket. -# Sending half of the pair that produced the Socket / TCP rows in -# docs/benchmarks/performance-dgx-spark.md. Run the server half from -# daqiri_bench_socket_tcp_server_spark_xhost.yaml on the peer host. -# -# Unlike the RoCE cross-host config, the socket engine binds every interface -# listed in the file during daqiri_init, so the two roles CANNOT share one -# config: the client host does not own the server's address and the bind would -# fail. Hence one file per role. -# -# Replace , , and the CPU placeholders for each host -# before running. Keep the same endpoint values in the peer server config. -# -# Before running, apply cross-host L3 on both boxes: -# sudo scripts/setup_spark_xhost_net.sh --role tx --peer-mac -# See docs/tutorials/system_configuration.md#cross-host-variant-two-sparks. -# -# Start the server first, then run the client. TCP self-paces through flow -# control, so it runs unthrottled -- no --target-gbps: -# sudo ./daqiri_bench_socket --mode client --seconds 30 -# -# Reproduces the 1 MiB row as shipped; for the 1000 B and 8000 B rows set -# message_size here AND in the server config. The buffers are sized for the -# 1 MiB message, which covers the smaller ones too. Note that message_size -# changes how many send() calls the app makes, not what the NIC transmits -- -# segmentation offload puts MTU-sized frames on the wire either way. -# -# `socket_bench_client.cpu_core` pins the benchmark worker. The socket engine -# does not currently apply queue `cpu_core` to its I/O threads. -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "socket" - master_core: - debug: false - log_level: "info" - - memory_regions: - - name: "DATA_SOCKET_CLIENT" - kind: "host" - affinity: 0 - num_bufs: 64 - buf_size: 1052672 - - interfaces: - - name: tcp_client - address: - socket_config: - mode: client - local_addr: "tcp://:6002" - remote_addr: "tcp://:6001" - rx: - queues: - - name: "Client_RX_Queue" - id: 0 - cpu_core: - batch_size: 1 - memory_regions: - - "DATA_SOCKET_CLIENT" - tx: - queues: - - name: "Client_TX_Queue" - id: 0 - cpu_core: - batch_size: 1 - memory_regions: - - "DATA_SOCKET_CLIENT" - -socket_bench_client: - server: false - send: true - receive: false - cpu_core: - iterations: 1000000000 - message_size: 1048576 - server_address: - client_address: - server_port: 6001 diff --git a/examples/daqiri_bench_socket_tcp_server_spark_xhost.yaml b/examples/daqiri_bench_socket_tcp_server_spark_xhost.yaml deleted file mode 100644 index eb89c223..00000000 --- a/examples/daqiri_bench_socket_tcp_server_spark_xhost.yaml +++ /dev/null @@ -1,77 +0,0 @@ -# DGX Spark (GB10) cross-host TCP server config for daqiri_bench_socket. -# Receiving half of the pair that produced the Socket / TCP rows in -# docs/benchmarks/performance-dgx-spark.md. Run the client half from -# daqiri_bench_socket_tcp_client_spark_xhost.yaml on the peer host. -# -# Unlike the RoCE cross-host config, the socket engine binds every interface -# listed in the file during daqiri_init, so the two roles CANNOT share one -# config: the server host does not own the client's address and the bind would -# fail. Hence one file per role. -# -# Replace , , and the CPU placeholders for each host -# before running. Keep the same endpoint values in the peer client config. -# -# Before running, apply cross-host L3 on both boxes: -# sudo scripts/setup_spark_xhost_net.sh --role rx --peer-mac -# See docs/tutorials/system_configuration.md#cross-host-variant-two-sparks. -# -# Run the server first and let it outlive the client: -# sudo ./daqiri_bench_socket --mode server --seconds 42 -# -# TCP self-paces through flow control, so it runs unthrottled -- no -# --target-gbps. Reproduces the 1 MiB row as shipped; for the 1000 B and 8000 B -# rows set message_size here AND in the client config. The buffers are sized -# for the 1 MiB message, which covers the smaller ones too. -# -# `socket_bench_server.cpu_core` pins the benchmark worker. The socket engine -# does not currently apply queue `cpu_core` to its I/O threads. -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "socket" - master_core: - debug: false - log_level: "info" - - memory_regions: - - name: "DATA_SOCKET_SERVER" - kind: "host" - affinity: 0 - num_bufs: 64 - buf_size: 1052672 - - interfaces: - - name: tcp_server - address: - socket_config: - mode: server - local_addr: "tcp://:6001" - rx: - queues: - - name: "Server_RX_Queue" - id: 0 - cpu_core: - batch_size: 1 - memory_regions: - - "DATA_SOCKET_SERVER" - tx: - queues: - - name: "Server_TX_Queue" - id: 0 - cpu_core: - batch_size: 1 - memory_regions: - - "DATA_SOCKET_SERVER" - -socket_bench_server: - server: true - send: false - receive: true - cpu_core: - iterations: 1000000000 - message_size: 1048576 - server_address: - server_port: 6001 diff --git a/examples/daqiri_bench_socket_tcp_tx_rx_spark_netns.yaml b/examples/daqiri_bench_socket_tcp_tx_rx_spark_netns.yaml deleted file mode 100644 index d03f87a7..00000000 --- a/examples/daqiri_bench_socket_tcp_tx_rx_spark_netns.yaml +++ /dev/null @@ -1,113 +0,0 @@ -# DGX Spark (GB10) TCP socket bench -- combined base for the netns wire loopback. -# -# This is the single checked-in config for the TCP wire-loopback sweep. It carries -# BOTH roles (the tcp_server / tcp_client interfaces, their memory regions, and the -# socket_bench_server / socket_bench_client sections). A single process can't run -# it as-is across namespaces (it would init the peer namespace's IP), so -# scripts/gen_spark_netns_config.py splits it to one role at sweep time and -# run_spark_bench.sh runs each role in its own namespace, assigning ports/cores per -# pair (cores 16-19 across four pairs; the send and receive sides of a pair share -# one core and self-pace, App TX ~= App RX). To split by hand: -# scripts/gen_spark_netns_config.py examples/daqiri_bench_socket_tcp_tx_rx_spark_netns.yaml \ -# --role server > server.yaml -# ip netns exec dq_wire_server daqiri_bench_socket server.yaml --seconds N --mode server -# -# Derived from examples/daqiri_bench_socket_tcp_tx_rx.yaml (the 127.0.0.1 loopback -# smoke config): wire IPs 10.250.0.1/2 and iterations bumped to 1e9 so a timed -# --seconds run does not exhaust early. buf_size/max_payload_size are 1 MiB: the -# bench memsets a full message_size (set per-cell, up to 1 MiB) into one TX buffer, -# so buf_size must be >= the largest message or the write overflows the heap; -# num_bufs is trimmed to keep the pool bounded. -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "socket" - master_core: 8 - debug: false - log_level: "info" - - memory_regions: - - name: "DATA_SOCKET_SERVER" - kind: "host" - affinity: 0 - num_bufs: 64 - buf_size: 1048576 - - name: "DATA_SOCKET_CLIENT" - kind: "host" - affinity: 0 - num_bufs: 64 - buf_size: 1048576 - - interfaces: - - name: tcp_server - address: 10.250.0.2 - socket_config: - mode: server - local_addr: "tcp://10.250.0.2:6001" - max_payload_size: 1048576 - rx: - queues: - - name: "Server_RX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - memory_regions: - - "DATA_SOCKET_SERVER" - tx: - queues: - - name: "Server_TX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - memory_regions: - - "DATA_SOCKET_SERVER" - - name: tcp_client - address: 10.250.0.1 - socket_config: - mode: client - local_addr: "tcp://10.250.0.1:6002" - remote_addr: "tcp://10.250.0.2:6001" - max_payload_size: 1048576 - tx: - queues: - - name: "Client_TX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - memory_regions: - - "DATA_SOCKET_CLIENT" - rx: - queues: - - name: "Client_RX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - memory_regions: - - "DATA_SOCKET_CLIENT" - -socket_bench_server: - # Bench worker thread affinity (PR #149). run_spark_bench.sh rewrites this to the - # pair core, so the server worker, client worker, and both queues share one core - # per pair -- the deliberate self-pacing setup (App TX ~= App RX). - cpu_core: 16 - server: true - send: false - receive: true - iterations: 1000000000 - message_size: 1024 - server_address: 10.250.0.2 - server_port: 6001 - -socket_bench_client: - cpu_core: 16 - server: false - send: true - receive: false - iterations: 1000000000 - message_size: 1024 - server_address: 10.250.0.2 - client_address: 10.250.0.1 - server_port: 6001 diff --git a/examples/daqiri_bench_socket_udp_client_spark_xhost.yaml b/examples/daqiri_bench_socket_udp_client_spark_xhost.yaml deleted file mode 100644 index 4827292a..00000000 --- a/examples/daqiri_bench_socket_udp_client_spark_xhost.yaml +++ /dev/null @@ -1,86 +0,0 @@ -# DGX Spark (GB10) cross-host UDP client config for daqiri_bench_socket. -# Sending half of the pair that produced the Socket / UDP rows in -# docs/benchmarks/performance-dgx-spark.md. Run the server half from -# daqiri_bench_socket_udp_server_spark_xhost.yaml on the peer host. -# -# Unlike the RoCE cross-host config, the socket engine binds every interface -# listed in the file during daqiri_init, so the two roles CANNOT share one -# config: the client host does not own the server's address and the bind would -# fail. Hence one file per role. -# -# Replace , , and the CPU placeholders for each host -# before running. Keep the same endpoint values in the peer server config. -# -# Before running, apply cross-host L3 on both boxes: -# sudo scripts/setup_spark_xhost_net.sh --role tx --peer-mac -# See docs/tutorials/system_configuration.md#cross-host-variant-two-sparks. -# -# Start the server first, then run the client. UDP has no flow control, so pace -# it: --target-gbps drives a token-bucket pacer, and the published rows are the -# highest rate that held zero loss over three reps (25 Gb/s at 8000 B). -# sudo ./daqiri_bench_socket --mode client --seconds 30 --target-gbps 25 -# Omit --target-gbps to measure the unpaced rate, where the receiver's drain -# rate decides what arrives. Walk the target up until the server's recv_bytes -# stops tracking the client's sent_bytes to find the loss-free rate for a -# different message size or core layout. -# -# Reproduces the 8000 B row as shipped. For the other published message sizes -# set message_size to 1000 or 65507 here AND in the server config; the buffers -# are already sized for the largest of them. -# -# `socket_bench_client.cpu_core` pins the benchmark worker. For UDP, -# `rx.queues[].cpu_core` pins the separate socket receive-I/O thread. Keep an -# RX I/O/worker pair in one performance cluster when measuring receive rate. -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "socket" - master_core: - debug: false - log_level: "info" - - memory_regions: - - name: "DATA_SOCKET_CLIENT" - kind: "host" - affinity: 0 - num_bufs: 1024 - buf_size: 65536 - - interfaces: - - name: udp_client - address: - socket_config: - mode: client - local_addr: "udp://:5002" - remote_addr: "udp://:5001" - max_payload_size: 65535 - rx: - queues: - - name: "Client_RX_Queue" - id: 0 - cpu_core: - batch_size: 1 - memory_regions: - - "DATA_SOCKET_CLIENT" - tx: - queues: - - name: "Client_TX_Queue" - id: 0 - cpu_core: - batch_size: 1 - memory_regions: - - "DATA_SOCKET_CLIENT" - -socket_bench_client: - server: false - send: true - receive: false - cpu_core: - iterations: 1000000000 - message_size: 8000 - server_address: - client_address: - server_port: 5001 diff --git a/examples/daqiri_bench_socket_udp_server_spark_xhost.yaml b/examples/daqiri_bench_socket_udp_server_spark_xhost.yaml deleted file mode 100644 index 9b6446ef..00000000 --- a/examples/daqiri_bench_socket_udp_server_spark_xhost.yaml +++ /dev/null @@ -1,80 +0,0 @@ -# DGX Spark (GB10) cross-host UDP server config for daqiri_bench_socket. -# Receiving half of the pair that produced the Socket / UDP rows in -# docs/benchmarks/performance-dgx-spark.md. Run the client half from -# daqiri_bench_socket_udp_client_spark_xhost.yaml on the peer host. -# -# Unlike the RoCE cross-host config, the socket engine binds every interface -# listed in the file during daqiri_init, so the two roles CANNOT share one -# config: the server host does not own the client's address and the bind would -# fail. Hence one file per role. -# -# Replace , , and the CPU placeholders for each host -# before running. Keep the same endpoint values in the peer client config. -# -# Before running, apply cross-host L3 on both boxes: -# sudo scripts/setup_spark_xhost_net.sh --role rx --peer-mac -# See docs/tutorials/system_configuration.md#cross-host-variant-two-sparks. -# -# Run the server first and let it outlive the client, so it is receiving before -# the first datagram and still draining after the last: -# sudo ./daqiri_bench_socket --mode server --seconds 42 -# -# Reproduces the 8000 B row as shipped. For the other published message sizes -# set message_size to 1000 or 65507 here AND in the client config; the buffers -# are already sized for the largest of them. -# -# `socket_bench_server.cpu_core` pins the benchmark worker. For UDP, -# `rx.queues[].cpu_core` pins the separate socket receive-I/O thread. Keep an -# RX I/O/worker pair in one performance cluster when measuring receive rate. -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "socket" - master_core: - debug: false - log_level: "info" - - memory_regions: - - name: "DATA_SOCKET_SERVER" - kind: "host" - affinity: 0 - num_bufs: 1024 - buf_size: 65536 - - interfaces: - - name: udp_server - address: - socket_config: - mode: server - local_addr: "udp://:5001" - remote_addr: "udp://:5002" - max_payload_size: 65535 - rx: - queues: - - name: "Server_RX_Queue" - id: 0 - cpu_core: - batch_size: 32 - memory_regions: - - "DATA_SOCKET_SERVER" - tx: - queues: - - name: "Server_TX_Queue" - id: 0 - cpu_core: - batch_size: 1 - memory_regions: - - "DATA_SOCKET_SERVER" - -socket_bench_server: - server: true - send: false - receive: true - cpu_core: - iterations: 1000000000 - message_size: 8000 - server_address: - server_port: 5001 diff --git a/examples/daqiri_bench_socket_udp_tx_rx_spark_netns.yaml b/examples/daqiri_bench_socket_udp_tx_rx_spark_netns.yaml deleted file mode 100644 index 7cfbeecd..00000000 --- a/examples/daqiri_bench_socket_udp_tx_rx_spark_netns.yaml +++ /dev/null @@ -1,115 +0,0 @@ -# DGX Spark (GB10) UDP socket bench -- combined base for the netns wire loopback. -# -# This is the single checked-in config for the UDP wire-loopback sweep. It carries -# BOTH roles (the udp_server / udp_client interfaces, their memory regions, and the -# socket_bench_server / socket_bench_client sections). A single process can't run -# it as-is across namespaces (it would init the peer namespace's IP), so -# scripts/gen_spark_netns_config.py splits it to one role at sweep time and -# run_spark_bench.sh runs each role in its own namespace, assigning ports/cores per -# pair. Server workers use [16, 18, 5, 7], client workers use [17, 19, 6, 9], and -# the server UDP I/O thread shares its server-worker core by default. -# SOCKET_RX_IO_CORES overrides the server I/O placement independently. To split by hand: -# scripts/gen_spark_netns_config.py examples/daqiri_bench_socket_udp_tx_rx_spark_netns.yaml \ -# --role server > server.yaml -# ip netns exec dq_wire_server daqiri_bench_socket server.yaml --seconds N --mode server -# -# Derived from examples/daqiri_bench_socket_udp_tx_rx.yaml (the 127.0.0.1 loopback -# smoke config): wire IPs 10.250.0.1/2 and iterations bumped to 1e9 so a timed -# --seconds run does not exhaust early. buf_size/max_payload_size are 65536/65535 -# (>= the 65507 max UDP datagram the sweep sends): the bench memsets a full -# message_size into one TX buffer, so a smaller buf_size overflows the heap. -# -%YAML 1.2 ---- -daqiri: - cfg: - version: 1 - stream_type: "socket" - master_core: 8 - debug: false - log_level: "info" - - memory_regions: - - name: "DATA_SOCKET_SERVER" - kind: "host" - affinity: 0 - num_bufs: 1024 - buf_size: 65536 - - name: "DATA_SOCKET_CLIENT" - kind: "host" - affinity: 0 - num_bufs: 1024 - buf_size: 65536 - - interfaces: - - name: udp_server - address: 10.250.0.2 - socket_config: - mode: server - local_addr: "udp://10.250.0.2:5001" - remote_addr: "udp://10.250.0.1:5002" - max_payload_size: 65535 - rx: - queues: - - name: "Server_RX_Queue" - id: 0 - cpu_core: 16 - # The socket engine passes up to one recvmmsg batch per DAQIRI burst. - batch_size: 32 - memory_regions: - - "DATA_SOCKET_SERVER" - tx: - queues: - - name: "Server_TX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - memory_regions: - - "DATA_SOCKET_SERVER" - - name: udp_client - address: 10.250.0.1 - socket_config: - mode: client - local_addr: "udp://10.250.0.1:5002" - remote_addr: "udp://10.250.0.2:5001" - max_payload_size: 65535 - tx: - queues: - - name: "Client_TX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - memory_regions: - - "DATA_SOCKET_CLIENT" - rx: - queues: - - name: "Client_RX_Queue" - id: 0 - cpu_core: 16 - batch_size: 1 - memory_regions: - - "DATA_SOCKET_CLIENT" - -socket_bench_server: - # Bench worker thread affinity (PR #149). run_spark_bench.sh rewrites this to - # the pair's server-worker core, separately from the client worker and optional - # UDP receive I/O core. - cpu_core: 16 - server: true - send: false - receive: true - iterations: 1000000000 - message_size: 1024 - server_address: 10.250.0.2 - server_port: 5001 - -socket_bench_client: - cpu_core: 16 - server: false - send: true - receive: false - iterations: 1000000000 - message_size: 1024 - server_address: 10.250.0.2 - client_address: 10.250.0.1 - server_port: 5001 diff --git a/examples/run_spark_bench.sh b/examples/run_spark_bench.sh index 546b29f7..49993beb 100755 --- a/examples/run_spark_bench.sh +++ b/examples/run_spark_bench.sh @@ -75,8 +75,13 @@ fi SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)" BUILD_DIR="${DAQIRI_BUILD_DIR:-$SCRIPT_DIR/../build}" -# Splits a combined both-role netns base into a single-role config (rdma + socket). -NETNS_GEN="$SCRIPT_DIR/../scripts/gen_spark_netns_config.py" +# Shared production/benchmark configuration generator. It emits complete, +# independently runnable role configs from the cell's actual parameters. +CONFIG_GEN="$SCRIPT_DIR/../scripts/gen_daqiri_config.py" +if [[ ! -f "$CONFIG_GEN" ]]; then + echo "Configuration generator not found: $CONFIG_GEN" >&2 + exit 1 +fi TS="$(date -u +%Y%m%dT%H%M%SZ)" OUT_DIR="$SCRIPT_DIR/../bench-results/$TS-$BACKEND-$MODE" mkdir -p "$OUT_DIR" @@ -197,7 +202,6 @@ case "$BACKEND" in BATCHES_HEADLINE=(10240) PAIRS_SWEEP=(1) PAIRS_HEADLINE=(1) - BASE_YAML="$SCRIPT_DIR/daqiri_bench_raw_tx_rx_spark.yaml" BENCH_BIN="$BUILD_DIR/examples/daqiri_bench_raw_gpudirect" CPU_MASTER=8; CPU_TX=17; CPU_RX=18 : "${ETH_DST_ADDR:?ETH_DST_ADDR must be set for dpdk backend (cat /sys/class/net//address)}" @@ -218,11 +222,6 @@ case "$BACKEND" in BATCHES_HEADLINE=(1) PAIRS_SWEEP=(1) PAIRS_HEADLINE=(1) - # One combined base (both roles, netns IPs 10.250.0.1/2); generate_yaml splits - # it per role at run time so each process runs inside its own network namespace - # and RDMA-CM resolves addresses over the wire rather than short-cutting through - # the kernel's local routing table. - BASE_YAML="$SCRIPT_DIR/daqiri_bench_rdma_tx_rx_spark_netns.yaml" BENCH_BIN="$BUILD_DIR/examples/daqiri_bench_rdma" # One-way roles: the client (send-only) drives the TX-queue core 17; the server # (receive-only) runs its RX-queue poller AND bench worker on core 19. Measure @@ -257,11 +256,6 @@ case "$BACKEND" in PAIRS_SWEEP=(1 2 4) PAIRS_HEADLINE=(4) SRV_PORT_BASE=5001; CLI_PORT_BASE=5101 - # One combined base (both roles, netns IPs 10.250.0.1/2); generate_socket_yaml - # splits it per role at run time so each process runs inside its own network - # namespace and the kernel sends client->server over the wire instead of short- - # cutting same-host IPs through the loopback (lo) device. - BASE_YAML="$SCRIPT_DIR/daqiri_bench_socket_udp_tx_rx_spark_netns.yaml" BENCH_BIN="$BUILD_DIR/examples/daqiri_bench_socket" # Final pair-0 TX/RX attribution is derived from the structured pinning below. CPU_MASTER=8; CPU_TX=17; CPU_RX=16 @@ -278,8 +272,6 @@ case "$BACKEND" in PAIRS_SWEEP=(1 2 4) PAIRS_HEADLINE=(4) SRV_PORT_BASE=6001; CLI_PORT_BASE=6101 - # One combined base (both roles, netns IPs 10.250.0.1/2); see socket-udp note. - BASE_YAML="$SCRIPT_DIR/daqiri_bench_socket_tcp_tx_rx_spark_netns.yaml" BENCH_BIN="$BUILD_DIR/examples/daqiri_bench_socket" # Final pair-0 TX/RX attribution is derived from the structured pinning below. CPU_MASTER=8; CPU_TX=17; CPU_RX=16 @@ -437,17 +429,24 @@ phy_counter() { | awk -F'[: ]+' -v k="$key" '$2 == k { s += $3 } END { printf "%d", s+0 }' } -# Substitute payload / batch into the base YAML and write a temp config (dpdk + rdma; -# sockets use generate_socket_yaml so each pair gets unique ports/cores). +# Generate a complete config from the cell parameters (DPDK + RoCE; sockets use +# generate_socket_yaml so each concurrent pair gets unique ports/cores). generate_yaml() { local out="$1" payload="$2" batch="$3" case "$BACKEND" in dpdk) - sed -E \ - -e "s|^( *payload_size: ).*|\1$payload|" \ - -e "s|^( *batch_size: ).*|\1$batch|" \ - -e "s|<00:00:00:00:00:00>|$ETH_DST_ADDR|g" \ - "$BASE_YAML" > "$out" + # Keep packet buffers at the report's original 8064-byte capacity so + # payload size remains the only memory-layout variable in this sweep. + python3 "$CONFIG_GEN" raw-pair \ + --tx-address "$DPDK_TX_PCI" --rx-address "$DPDK_RX_PCI" \ + --master-core 8 --engine dpdk --memory-kind host_pinned \ + --tx-queue-cores 17 --rx-queue-cores 18 \ + --tx-worker-cores 16 --rx-worker-cores 19 \ + --payload-size "$payload" --buffer-size 8064 \ + --batch-size "$batch" --num-bufs 51200 \ + --eth-dst-addr "$ETH_DST_ADDR" \ + --ip-src-addr 1.1.1.1 --ip-dst-addr 2.2.2.2 \ + --output "$out" || return 1 ;; rdma) # Size the flow-control window per message size (PR #144). buf_size tracks the @@ -463,31 +462,30 @@ generate_yaml() { local cap=$(( budget / payload )); (( cap < 1 )) && cap=1 local rx_nb=${RDMA_RX_NB:-512}; (( rx_nb > cap )) && rx_nb=$cap local tx_nb=${RDMA_TX_NB:-128}; (( tx_nb > cap )) && tx_nb=$cap - # Split the combined base per role, then apply the per-message-size window - # rewrite to each. Server -> $out, client -> ${out%.yaml}_client.yaml. - local role dst - for role in server client; do - if [[ "$role" == server ]]; then dst="$out"; else dst="${out%.yaml}_client.yaml"; fi - # name-anchored num_bufs rewrite: RX regions -> rx_nb, TX regions -> tx_nb. - # Depths are clamped to their region's num_bufs so the window never exceeds - # the buffers backing it. One-way (server receive-only + GEMM, client - # send-only) is baked into the base config, matching the DPDK and socket - # benches -- see the send:/receive: notes in the base YAML. - python3 "$NETNS_GEN" "$BASE_YAML" --role "$role" | \ - awk -v p="$payload" -v bs="$payload" -v rxnb="$rx_nb" -v txnb="$tx_nb" ' - /^[[:space:]]*- name:/ { region = $0 } - /^[[:space:]]*num_bufs:/ { - if (region ~ /RX/) { sub(/num_bufs:.*/, "num_bufs: " rxnb) } - else if (region ~ /TX/) { sub(/num_bufs:.*/, "num_bufs: " txnb) } - print; next - } - /^[[:space:]]*buf_size:/ { sub(/buf_size:.*/, "buf_size: " bs); print; next } - /^[[:space:]]*message_size:/ { sub(/message_size:.*/, "message_size: " p); print; next } - /^[[:space:]]*rx_depth:/ { sub(/rx_depth:.*/, "rx_depth: " rxnb); print; next } - /^[[:space:]]*tx_depth:/ { sub(/tx_depth:.*/, "tx_depth: " txnb); print; next } - { print } - ' > "$dst" - done + # Direction-specific region counts and queue depths keep the flow-control + # window within the buffers that back it. Server -> $out; client -> sibling. + python3 "$CONFIG_GEN" socket-pair --transport roce \ + --client-address 10.250.0.1 --server-address 10.250.0.2 \ + --client-port 4096 --server-port 4096 \ + --client-master-core 8 --server-master-core 8 \ + --client-rx-core 18 --client-tx-core 17 \ + --server-rx-core 19 --server-tx-core 16 \ + --client-worker-core 18 --server-worker-core 19 \ + --message-size "$payload" --buffer-size "$payload" --num-bufs 1 \ + --rx-num-bufs "$rx_nb" --tx-num-bufs "$tx_nb" \ + --rx-depth "$rx_nb" --tx-depth "$tx_nb" --memory-kind host_pinned \ + --role rx --output "$out" || return 1 + python3 "$CONFIG_GEN" socket-pair --transport roce \ + --client-address 10.250.0.1 --server-address 10.250.0.2 \ + --client-port 4096 --server-port 4096 \ + --client-master-core 8 --server-master-core 8 \ + --client-rx-core 18 --client-tx-core 17 \ + --server-rx-core 19 --server-tx-core 16 \ + --client-worker-core 18 --server-worker-core 19 \ + --message-size "$payload" --buffer-size "$payload" --num-bufs 1 \ + --rx-num-bufs "$rx_nb" --tx-num-bufs "$tx_nb" \ + --rx-depth "$rx_nb" --tx-depth "$tx_nb" --memory-kind host_pinned \ + --role tx --output "${out%.yaml}_client.yaml" || return 1 ;; esac } @@ -536,10 +534,8 @@ if [[ "$BACKEND" =~ ^socket- ]]; then fi fi -# Write the server/client YAML pair for socket pair `idx`: split the combined base -# per role, then substitute message_size, unique ports (SRV/CLI_PORT_BASE + idx), -# and pin the server (receive) side and client (send) side to DIFFERENT isolated -# cores so the pair does not time-slice one CPU (see pair_server_core comment). +# Write a complete server/client YAML pair with unique ports and independent +# server/client placement, so the pair does not time-slice one CPU. generate_socket_yaml() { local idx="$1" payload="$2" batch="$3" server_out="$4" client_out="$5" local srv_port=$(( SRV_PORT_BASE + idx )) @@ -558,25 +554,27 @@ generate_socket_yaml() { server_io_core="$(pair_server_io_core "$idx" "$server_core")" fi fi - python3 "$NETNS_GEN" "$BASE_YAML" --role server \ - --rx-queue-cpu-core "$server_io_core" --rx-queue-batch-size "$batch" \ - --tx-queue-cpu-core "$server_core" \ - --bench-cpu-core "$server_core" | \ - sed -E \ - -e "s|^( *message_size: ).*|\1$payload|g" \ - -e "s|^( *local_addr: \"?[a-z]+://[0-9.]+:)[0-9]+(\"?)|\1$srv_port\2|" \ - -e "s|^( *remote_addr: \"?[a-z]+://[0-9.]+:)[0-9]+(\"?)|\1$cli_port\2|" \ - -e "s|^( *server_port: ).*|\1$srv_port|" \ - > "$server_out" - python3 "$NETNS_GEN" "$BASE_YAML" --role client \ - --rx-queue-cpu-core "$client_core" --tx-queue-cpu-core "$client_core" \ - --bench-cpu-core "$client_core" | \ - sed -E \ - -e "s|^( *message_size: ).*|\1$payload|g" \ - -e "s|^( *local_addr: \"?[a-z]+://[0-9.]+:)[0-9]+(\"?)|\1$cli_port\2|" \ - -e "s|^( *remote_addr: \"?[a-z]+://[0-9.]+:)[0-9]+(\"?)|\1$srv_port\2|" \ - -e "s|^( *server_port: ).*|\1$srv_port|" \ - > "$client_out" + local transport="${BACKEND#socket-}" + local buffer_size=65536 num_bufs=1024 + if [[ "$transport" == "tcp" ]]; then + buffer_size=1048576 + num_bufs=64 + fi + local common_args=( + --transport "$transport" + --client-address 10.250.0.1 --server-address 10.250.0.2 + --client-port "$cli_port" --server-port "$srv_port" + --client-master-core 8 --server-master-core 8 + --client-rx-core "$client_core" --client-tx-core "$client_core" + --server-rx-core "$server_io_core" --server-tx-core "$server_core" + --client-worker-core "$client_core" --server-worker-core "$server_core" + --message-size "$payload" --buffer-size "$buffer_size" + --num-bufs "$num_bufs" --rx-batch-size "$batch" + ) + python3 "$CONFIG_GEN" socket-pair "${common_args[@]}" --role rx \ + --output "$server_out" || return 1 + python3 "$CONFIG_GEN" socket-pair "${common_args[@]}" --role tx \ + --output "$client_out" || return 1 } # Run one cell. Echoes the CSV row to stdout. @@ -586,6 +584,26 @@ run_cell() { local cell_dir="$OUT_DIR/$cell" mkdir -p "$cell_dir" + # Generate every input before starting monitors or benchmark processes. A + # failed generator must fail this cell rather than launching with missing or + # partially written YAML. + local yaml="" i + if [[ "$BACKEND" =~ ^socket- ]]; then + for ((i = 0; i < pairs; i++)); do + if ! generate_socket_yaml "$i" "$payload" "$batch" \ + "$cell_dir/server_p$i.yaml" "$cell_dir/client_p$i.yaml"; then + echo "ERROR: $cell configuration generation failed for socket pair $i" >&2 + return 1 + fi + done + else + yaml="$cell_dir/config.yaml" + if ! generate_yaml "$yaml" "$payload" "$batch"; then + echo "ERROR: $cell configuration generation failed" >&2 + return 1 + fi + fi + # Snapshot kernel-side drop counters. In the netns wire loopback the UDP # receiver lives in the server netns and TCP retransmits are counted on the # client (sender) netns, so read each counter inside the relevant namespace. @@ -633,11 +651,7 @@ run_cell() { # namespaces with unique ports and cores. A single pair is core-bound below line # rate; the published Spark matrix scales aggregate throughput with four pairs. # App TX (client sent) and App RX (server recv) are summed across pairs. - local i server_pids=() client_pids=() - for ((i = 0; i < pairs; i++)); do - generate_socket_yaml "$i" "$payload" "$batch" \ - "$cell_dir/server_p$i.yaml" "$cell_dir/client_p$i.yaml" - done + local server_pids=() client_pids=() for ((i = 0; i < pairs; i++)); do ip netns exec dq_wire_server "$BENCH_BIN" "$cell_dir/server_p$i.yaml" \ --seconds "$server_seconds" "${bench_extra[@]}" --mode server \ @@ -677,8 +691,6 @@ run_cell() { elif [[ "$BACKEND" == "rdma" ]]; then # Split server/client processes in separate namespaces so RDMA-CM resolves # addresses over the wire rather than short-cutting the kernel's local table. - local yaml="$cell_dir/config.yaml" - generate_yaml "$yaml" "$payload" "$batch" # Optionally wrap the server (receive + GPU-workload) in nsys. Placed AFTER the # netns exec so nsys traces the bench, not `ip netns exec`. local -a nsys_pre=() @@ -737,8 +749,6 @@ run_cell() { echo "INFO: $cell wire OK -- client tx_packets_phy +$phy_tx_delta, server rx_packets_phy +$phy_rx_delta (>= $phy_min msgs)" >&2 fi else - local yaml="$cell_dir/config.yaml" - generate_yaml "$yaml" "$payload" "$batch" # Snapshot the p0/p1 *_phy counters around the run to assert the packets crossed # the cable (tx_port -> rx_port over the wire), not the on-chip eswitch short-cut. local phy_tx_before phy_rx_before diff --git a/examples/run_spark_mq_bench.sh b/examples/run_spark_mq_bench.sh index 88a61f91..e04af7fb 100755 --- a/examples/run_spark_mq_bench.sh +++ b/examples/run_spark_mq_bench.sh @@ -17,10 +17,8 @@ # 2t1r 16,19 18 # 2t2r 16,19 18,9 # -# All four are derived from the single checked-in base -# examples/daqiri_bench_raw_tx_rx_spark_mq.yaml (the balanced 2,2 superset) by -# scripts/gen_spark_mq_config.py, which prunes queues/flows/memory-regions/bench -# entries down to each cell. They share host_pinned memory, an over-the-wire +# All four are emitted directly by scripts/gen_daqiri_config.py from the cell's +# queue counts, placement, and host bindings. They share host_pinned memory, an over-the-wire # loopback (tx 0000:01:00.0 -> rx 0002:01:00.1), and master_core 8. The native # shape is an 8000 B payload; this script sweeps PAYLOADS (default 64..8000 B), # generating a fresh config per (cell, payload) -- it NEVER edits the base. @@ -79,9 +77,8 @@ PAYLOADS="${PAYLOADS:-64 256 1024 4096 8000}" # reps. Default 1; set REPEATS=3 for the published re-run. REPEATS="${REPEATS:-1}" -# Single checked-in base + the generator that prunes it to each cell. -MQ_BASE="$SCRIPT_DIR/daqiri_bench_raw_tx_rx_spark_mq.yaml" -MQ_GEN="$SCRIPT_DIR/../scripts/gen_spark_mq_config.py" +# Shared production/benchmark configuration generator. +CONFIG_GEN="$SCRIPT_DIR/../scripts/gen_daqiri_config.py" # Resolve the DAQIRI shared libs from the build tree first. The per-engine # sub-libraries (libdaqiri_dpdk.so.0 etc.) live in $BUILD_DIR/src, not $BUILD_DIR, @@ -109,8 +106,8 @@ if [[ ! -x "$BENCH_BIN" ]]; then exit 1 fi -if [[ ! -f "$MQ_BASE" || ! -f "$MQ_GEN" ]]; then - echo "ERROR: multi-queue base/generator missing: $MQ_BASE / $MQ_GEN" >&2 +if [[ ! -f "$CONFIG_GEN" ]]; then + echo "ERROR: config generator missing: $CONFIG_GEN" >&2 exit 1 fi @@ -122,6 +119,14 @@ RX_NETDEV="${RX_NETDEV:-}" if [[ -z "$RX_NETDEV" ]]; then RX_NETDEV="$(ls "/sys/bus/pci/devices/$RX_PCI/net" 2>/dev/null | head -n1 || true)" fi +ETH_DST_ADDR="${ETH_DST_ADDR:-}" +if [[ -z "$ETH_DST_ADDR" && -n "$RX_NETDEV" ]]; then + ETH_DST_ADDR="$(cat "/sys/class/net/$RX_NETDEV/address" 2>/dev/null || true)" +fi +if [[ -z "$ETH_DST_ADDR" ]]; then + echo "ERROR: could not resolve the RX-port MAC; set ETH_DST_ADDR" >&2 + exit 1 +fi # cell name -> "tx_queue_count rx_queue_count". The CSV's tx_cores/rx_cores # display columns are derived from these counts (TX -> cores 16[,17], RX -> @@ -223,13 +228,23 @@ run_cell() { local run_dir="$OUT_DIR/$cell/p$payload/r$rep" mkdir -p "$run_dir" - # Generate the cell from the single base -- never touch the base. Fill the - # rx_port MAC from ETH_DST_ADDR when set (the RX queue runs flow_isolation). + # Generate the complete cell directly from its topology and placement. local tmp_cfg="$run_dir/config.yaml" - local eth_dst_arg=() - [[ -n "${ETH_DST_ADDR:-}" ]] && eth_dst_arg=(--eth-dst "$ETH_DST_ADDR") - if ! python3 "$MQ_GEN" "$MQ_BASE" --tx "$tx_count" --rx "$rx_count" \ - --payload "$payload" "${eth_dst_arg[@]}" > "$tmp_cfg" 2> "$run_dir/gen.err"; then + local tx_queue_cores="16" tx_worker_cores="15" + local rx_queue_cores="18" rx_worker_cores="17" + [[ "$tx_count" == 2 ]] && tx_queue_cores="16,19" && tx_worker_cores="15,6" + [[ "$rx_count" == 2 ]] && rx_queue_cores="18,9" && rx_worker_cores="17,7" + # Preserve the published sweep methodology: payload varies, while every + # packet buffer retains the original base configuration's 8064-byte capacity. + if ! python3 "$CONFIG_GEN" raw-pair \ + --tx-address 0000:01:00.0 --rx-address 0002:01:00.1 \ + --master-core 8 --engine dpdk --memory-kind host_pinned \ + --tx-queue-cores "$tx_queue_cores" --rx-queue-cores "$rx_queue_cores" \ + --tx-worker-cores "$tx_worker_cores" --rx-worker-cores "$rx_worker_cores" \ + --payload-size "$payload" --buffer-size 8064 \ + --batch-size 10240 --num-bufs 51200 \ + --eth-dst-addr "$ETH_DST_ADDR" --ip-src-addr 1.1.1.1 --ip-dst-addr 2.2.2.2 \ + > "$tmp_cfg" 2> "$run_dir/gen.err"; then echo "ERROR: $cell p$payload config generation failed" >&2 cat "$run_dir/gen.err" >&2 return 1 diff --git a/mkdocs.yml b/mkdocs.yml index 7275c3c1..7208325a 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -63,6 +63,7 @@ nav: - API Reference: - API Guide: api-reference/index.md - Configuration YAML Reference: api-reference/configuration.md + - Configuration Generation: config-generation.md - C++ API Usage: api-reference/cpp.md - Python API Usage: api-reference/python.md - Tutorials: diff --git a/scripts/check_daqiri_configs.py b/scripts/check_daqiri_configs.py index e3942774..8a4b5a4f 100755 --- a/scripts/check_daqiri_configs.py +++ b/scripts/check_daqiri_configs.py @@ -14,6 +14,14 @@ import tempfile from pathlib import Path +from config_validation import ( + ENGINE_QUERY_ERROR, + NO_SUPPORTED_CONFIGURATIONS, + query_compiled_engines, + required_engines_from_text, + select_supported_paths, +) + REPOSITORY_ROOT = Path(__file__).resolve().parents[1] INTEGER_PLACEHOLDER = re.compile( @@ -24,16 +32,6 @@ "": "192.0.2.1", "": "192.0.2.2", } -DEFAULT_CONFIGS = ( - "examples/daqiri_bench_raw_sw_loopback.yaml", - "examples/daqiri_bench_raw_tx_rx.yaml", - "examples/daqiri_bench_raw_tx_rx_hds.yaml", - "examples/daqiri_bench_rdma_tx_rx.yaml", - "examples/daqiri_bench_socket_udp_tx_rx.yaml", - "examples/daqiri_bench_socket_tcp_tx_rx.yaml", - "examples/daqiri_example_dynamic_rx_flow.yaml", - "examples/daqiri_example_named_endpoints_tx_rx.yaml", -) SEMANTIC_FIXTURE = "examples/daqiri_bench_raw_rx_reorder_seq_batch.yaml" ZERO_ID_FIXTURE = "examples/daqiri_bench_raw_tx_rx.yaml" PARSER_FIXTURE = "examples/daqiri_bench_socket_udp_tx_rx.yaml" @@ -71,7 +69,11 @@ def materialize_integer_placeholders(text: str) -> str: def checked_in_paths() -> list[Path]: - return [REPOSITORY_ROOT / relative_path for relative_path in DEFAULT_CONFIGS] + paths = sorted((REPOSITORY_ROOT / "examples").glob("daqiri_*.yaml")) + paths.extend( + sorted((REPOSITORY_ROOT / "applications").glob("**/configs/*.yaml")) + ) + return paths def zero_id_multi_interface_case() -> str: @@ -216,6 +218,18 @@ def main(argv: list[str] | None = None) -> int: if not paths: parser.error("no configuration files were found") + available_engines: frozenset[str] | None = None + if not args.paths: + try: + available_engines = query_compiled_engines(args.validator) + except RuntimeError: + print(ENGINE_QUERY_ERROR, file=sys.stderr) + return 1 + paths = select_supported_paths(paths, available_engines) + if not paths: + print(NO_SUPPORTED_CONFIGURATIONS, file=sys.stderr) + return 1 + compatibility_count = 0 invalid_count = 0 with tempfile.TemporaryDirectory(prefix="daqiri-checked-configs-") as temp_dir: @@ -246,21 +260,30 @@ def main(argv: list[str] | None = None) -> int: except (OSError, ValueError) as error: print(error, file=sys.stderr) return 1 - compatibility_path = Path(temp_dir) / "zero-flow-id-per-interface.yaml" - compatibility_path.write_text(compatibility_case, encoding="utf-8") - result = subprocess.run( - [str(args.validator), str(compatibility_path)], check=False - ) - if result.returncode != 0: - print( - "zero-flow-id-per-interface.yaml: expected validator exit status 0, " - f"got {result.returncode}", - file=sys.stderr, + if "dpdk" in available_engines: + compatibility_path = Path(temp_dir) / "zero-flow-id-per-interface.yaml" + compatibility_path.write_text(compatibility_case, encoding="utf-8") + result = subprocess.run( + [str(args.validator), str(compatibility_path)], check=False ) - return 1 - compatibility_count = 1 - invalid_count = len(invalid_cases) - for name, text in invalid_cases.items(): + if result.returncode != 0: + print( + "zero-flow-id-per-interface.yaml: expected validator exit status 0, " + f"got {result.returncode}", + file=sys.stderr, + ) + return 1 + compatibility_count = 1 + supported_invalid_cases = { + name: text + for name, text in invalid_cases.items() + if ( + (required := required_engines_from_text(text)) is None + or required & available_engines + ) + } + invalid_count = len(supported_invalid_cases) + for name, text in supported_invalid_cases.items(): invalid_path = Path(temp_dir) / name invalid_path.write_text(text, encoding="utf-8") result = subprocess.run( diff --git a/scripts/check_generated_configs.py b/scripts/check_generated_configs.py new file mode 100755 index 00000000..572ac41d --- /dev/null +++ b/scripts/check_generated_configs.py @@ -0,0 +1,201 @@ +#!/usr/bin/env python3 +# +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Exercise the generated configuration matrix with the C++ validator.""" + +from __future__ import annotations + +import argparse +import os +import subprocess +import sys +import tempfile +from pathlib import Path +from typing import Any + +from config_validation import ( + ENGINE_QUERY_ERROR, + NO_SUPPORTED_CONFIGURATIONS, + query_compiled_engines, + supports_engines, +) +from daqiri_config import ( + RawPairSpec, + SocketPairSpec, + generate_raw_pair, + generate_raw_roles, + generate_socket_pair, + render_document, +) + + +def _socket_spec(transport: str) -> SocketPairSpec: + message_size = {"udp": 8000, "tcp": 1_048_576, "roce": 8_000_000}[transport] + return SocketPairSpec( + transport=transport, + client_address="10.250.0.1", + server_address="10.250.0.2", + client_port=5002, + server_port=5001, + client_master_core=8, + server_master_core=8, + client_rx_core=18, + client_tx_core=17, + server_rx_core=19, + server_tx_core=16, + client_worker_core=18, + server_worker_core=19, + message_size=message_size, + buffer_size=message_size if transport != "udp" else 65536, + num_bufs=128, + rx_num_bufs=512 if transport == "roce" else None, + tx_num_bufs=128 if transport == "roce" else None, + rx_batch_size=32 if transport == "udp" else None, + memory_kind="host_pinned" if transport == "roce" else "host", + rx_depth=512 if transport == "roce" else None, + tx_depth=128 if transport == "roce" else None, + ) + + +def _raw_spec(**overrides: Any) -> RawPairSpec: + values: dict[str, Any] = { + "tx_address": "0000:01:00.0", + "rx_address": "0000:01:00.1", + "master_core": 3, + "tx_queue_cores": (4,), + "rx_queue_cores": (5,), + "tx_worker_cores": (6,), + "rx_worker_cores": (7,), + "eth_dst_addr": "02:00:00:00:00:02", + "eth_src_addr": "02:00:00:00:00:01", + "engine": "ibverbs", + } + values.update(overrides) + return RawPairSpec(**values) + + +def generated_matrix() -> dict[str, dict[str, Any]]: + documents: dict[str, dict[str, Any]] = {} + for transport in ("udp", "tcp", "roce"): + for role, document in generate_socket_pair(_socket_spec(transport)).items(): + documents[f"socket-{transport}-{role}"] = document + + for engine in ("dpdk", "ibverbs"): + for transform in ("none", "vlan", "vxlan", "gre", "nvgre"): + documents[f"raw-{engine}-{transform}"] = generate_raw_pair( + _raw_spec(engine=engine, transform=transform) + ) + + for tx_count, rx_count in ((1, 1), (1, 2), (2, 1), (2, 2)): + documents[f"raw-mq-{tx_count}x{rx_count}"] = generate_raw_pair( + _raw_spec( + engine="dpdk", + memory_kind="host_pinned", + tx_queue_cores=tuple(range(10, 10 + tx_count)), + rx_queue_cores=tuple(range(20, 20 + rx_count)), + tx_worker_cores=tuple(range(30, 30 + tx_count)), + rx_worker_cores=tuple(range(40, 40 + rx_count)), + ) + ) + + for role, document in generate_raw_roles(_raw_spec()).items(): + documents[f"raw-xhost-{role}"] = document + documents["raw-production-no-tx-offload"] = generate_raw_pair( + _raw_spec(include_benchmark=False, tx_eth_src=False, buffer_size=8064) + ) + return documents + + +def _explicit_engine(document: dict[str, Any]) -> str | None: + return document.get("daqiri", {}).get("cfg", {}).get("engine") + + +def _rendered_matrix() -> bytes: + output: list[str] = [] + for name, document in sorted(generated_matrix().items()): + output.append(f"# {name}\n") + output.append(render_document(document)) + return "".join(output).encode("utf-8") + + +def _generate_matrix_in_subprocess(hash_seed: int) -> bytes: + environment = os.environ.copy() + environment["PYTHONHASHSEED"] = str(hash_seed) + result = subprocess.run( + [sys.executable, str(Path(__file__).resolve()), "--emit-matrix"], + check=True, + capture_output=True, + env=environment, + ) + return result.stdout + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--validator", + type=Path, + help=( + "built daqiri_config_validate binary used for hardware-free " + "acceptance checks" + ), + ) + parser.add_argument( + "--exclude-dpdk", + action="store_true", + help="skip profiles that explicitly require DPDK", + ) + parser.add_argument("--emit-matrix", action="store_true", help=argparse.SUPPRESS) + args = parser.parse_args() + + if args.emit_matrix: + sys.stdout.buffer.write(_rendered_matrix()) + return 0 + if args.validator is None: + parser.error( + "--validator is required; configuration validation uses the C++ decoder" + ) + try: + available_engines = query_compiled_engines(args.validator) + except RuntimeError: + print(ENGINE_QUERY_ERROR, file=sys.stderr) + return 1 + + first_generation = _generate_matrix_in_subprocess(1) + second_generation = _generate_matrix_in_subprocess(2) + if first_generation != second_generation: + raise RuntimeError( + "non-deterministic generation across independent processes" + ) + + documents = { + name: document + for name, document in generated_matrix().items() + if (not args.exclude_dpdk or _explicit_engine(document) != "dpdk") + and supports_engines(document, available_engines) + } + if not documents: + print(NO_SUPPORTED_CONFIGURATIONS, file=sys.stderr) + return 1 + with tempfile.TemporaryDirectory(prefix="daqiri-generated-configs-") as temp_dir: + paths: dict[str, Path] = {} + for name, document in documents.items(): + rendered = render_document(document) + path = Path(temp_dir) / f"{name}.yaml" + path.write_text(rendered, encoding="utf-8") + paths[name] = path + + subprocess.run( + [str(args.validator), *(str(path) for path in paths.values())], check=True + ) + + print( + f"Validated {len(documents)} generated configurations with the C++ validator." + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/check_pr.sh b/scripts/check_pr.sh index 1a3c6a62..4dcbae01 100755 --- a/scripts/check_pr.sh +++ b/scripts/check_pr.sh @@ -16,6 +16,7 @@ Usage: scripts/check_pr.sh [--diagrams] [--docker-base] Runs the standard local checks before opening a PR: - portable Python tests - checked-in configuration validation + - generated configuration matrix validation - documentation build and documentation reference checks Set DAQIRI_CONFIG_VALIDATOR to the validator built in the project container. @@ -64,6 +65,7 @@ if [ ! -x "${DAQIRI_CONFIG_VALIDATOR}" ]; then exit 1 fi "${PYTHON}" scripts/check_daqiri_configs.py --validator "${DAQIRI_CONFIG_VALIDATOR}" +"${PYTHON}" scripts/check_generated_configs.py --validator "${DAQIRI_CONFIG_VALIDATOR}" docs_args=() if [ "${run_diagrams}" -eq 1 ]; then diff --git a/scripts/config_validation.py b/scripts/config_validation.py new file mode 100644 index 00000000..4ff947ec --- /dev/null +++ b/scripts/config_validation.py @@ -0,0 +1,125 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import subprocess +from collections.abc import Iterable, Mapping +from pathlib import Path +from typing import Any + +import yaml + + +KNOWN_ENGINES = frozenset(("socket", "dpdk", "ibverbs")) +ENGINE_QUERY_ERROR = "Cannot determine the validator's compiled engines." +NO_SUPPORTED_CONFIGURATIONS = "No configurations are supported by this validator." + + +def query_compiled_engines(validator: Path) -> frozenset[str]: + try: + result = subprocess.run( + [str(validator), "--list-engines"], + check=False, + capture_output=True, + text=True, + ) + except OSError as error: + raise RuntimeError from error + if result.returncode != 0: + raise RuntimeError + lines = result.stdout.splitlines() + if len(lines) != 1: + raise RuntimeError + engines = result.stdout.split() + if not engines or len(engines) != len(set(engines)): + raise RuntimeError + if not set(engines).issubset(KNOWN_ENGINES): + raise RuntimeError + return frozenset(engines) + + +def _config_mapping(document: Any) -> Mapping[str, Any] | None: + if not isinstance(document, Mapping): + return None + daqiri = document.get("daqiri") + if daqiri is not None: + if not isinstance(daqiri, Mapping) or not isinstance(daqiri.get("cfg"), Mapping): + return None + return daqiri["cfg"] + return document + + +def required_engines(document: Any) -> frozenset[str] | None: + config = _config_mapping(document) + if config is None: + return None + + stream_type = config.get("stream_type") + explicit_engine = config.get("engine") + if stream_type == "raw": + if explicit_engine in (None, "", "default"): + return frozenset(("dpdk", "ibverbs")) + if explicit_engine in ("dpdk", "ibverbs"): + return frozenset((explicit_engine,)) + return None + + if stream_type != "socket": + return None + if explicit_engine == "ibverbs": + return frozenset(("ibverbs",)) + if explicit_engine not in (None, "", "default", "socket"): + return None + + interfaces = config.get("interfaces") + if not isinstance(interfaces, Iterable) or isinstance(interfaces, (str, bytes)): + return None + protocols: set[str] = set() + for interface in interfaces: + if not isinstance(interface, Mapping): + return None + socket_config = interface.get("socket_config") + if not isinstance(socket_config, Mapping): + return None + addresses = [socket_config.get("local_addr"), socket_config.get("remote_addr")] + for address in addresses: + if not isinstance(address, str) or "://" not in address: + continue + protocols.add(address.split("://", 1)[0].lower()) + if not protocols or not protocols.issubset({"udp", "tcp", "roce"}): + return None + if "roce" in protocols: + return frozenset(("ibverbs",)) + return frozenset(("socket",)) + + +def required_engines_from_path(path: Path) -> frozenset[str] | None: + try: + text = path.read_text(encoding="utf-8") + except OSError: + return None + return required_engines_from_text(text) + + +def required_engines_from_text(text: str) -> frozenset[str] | None: + try: + document = yaml.safe_load(text) + except yaml.YAMLError: + return None + return required_engines(document) + + +def supports_engines(document: Any, available_engines: frozenset[str]) -> bool: + required = required_engines(document) + return required is None or bool(required & available_engines) + + +def select_supported_paths( + paths: Iterable[Path], available_engines: frozenset[str] +) -> list[Path]: + selected: list[Path] = [] + for path in paths: + required = required_engines_from_path(path) + if required is None or required & available_engines: + selected.append(path) + return selected diff --git a/scripts/daqiri_config/__init__.py b/scripts/daqiri_config/__init__.py new file mode 100644 index 00000000..3b3939af --- /dev/null +++ b/scripts/daqiri_config/__init__.py @@ -0,0 +1,30 @@ +# SPDX-FileCopyrightText: 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Deterministic DAQIRI configuration generation.""" + +from .core import ( + ConfigError, + apply_overrides, + load_document, + render_document, +) +from .profiles import ( + RawPairSpec, + SocketPairSpec, + generate_raw_pair, + generate_raw_roles, + generate_socket_pair, +) + +__all__ = [ + "ConfigError", + "RawPairSpec", + "SocketPairSpec", + "apply_overrides", + "generate_raw_pair", + "generate_raw_roles", + "generate_socket_pair", + "load_document", + "render_document", +] diff --git a/scripts/daqiri_config/core.py b/scripts/daqiri_config/core.py new file mode 100644 index 00000000..a7d55008 --- /dev/null +++ b/scripts/daqiri_config/core.py @@ -0,0 +1,152 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Deterministic YAML loading, overrides, and serialization.""" + +from __future__ import annotations + +import re +from pathlib import Path +from typing import Any + +import yaml + + +class ConfigError(ValueError): + """Raised when generator input cannot be loaded, overridden, or rendered.""" + + +class Yaml12SafeLoader(yaml.SafeLoader): + """Safe loader with YAML 1.2-style scalar resolution. + + In particular, YAML 1.1's sexagesimal float rule turns an unquoted PCI BDF + such as ``0000:01:00.0`` into ``60.0``. DAQIRI configs declare YAML 1.2, + where that scalar is a string. + """ + + +Yaml12SafeLoader.yaml_implicit_resolvers = { + key: list(resolvers) + for key, resolvers in yaml.SafeLoader.yaml_implicit_resolvers.items() +} +for first_character, resolvers in Yaml12SafeLoader.yaml_implicit_resolvers.items(): + Yaml12SafeLoader.yaml_implicit_resolvers[first_character] = [ + resolver + for resolver in resolvers + if resolver[0] + not in ( + "tag:yaml.org,2002:bool", + "tag:yaml.org,2002:float", + "tag:yaml.org,2002:int", + ) + ] +Yaml12SafeLoader.add_implicit_resolver( + "tag:yaml.org,2002:bool", re.compile(r"^(?:true|false)$", re.IGNORECASE), list("tTfF") +) +Yaml12SafeLoader.add_implicit_resolver( + "tag:yaml.org,2002:int", + re.compile(r"^[-+]?(?:0|[1-9][0-9]*|0o[0-7]+|0x[0-9a-fA-F]+)$"), + list("-+0123456789"), +) +Yaml12SafeLoader.add_implicit_resolver( + "tag:yaml.org,2002:float", + re.compile( + r"^[-+]?(?:(?:[0-9]+\.[0-9]*|\.[0-9]+)(?:[eE][-+]?[0-9]+)?|" + r"[0-9]+[eE][-+]?[0-9]+|\.inf|\.Inf|\.INF|\.nan|\.NaN|\.NAN)$" + ), + list("-+0123456789."), +) + + +def load_document(path: str | Path) -> dict[str, Any]: + """Load a YAML document from *path*.""" + + try: + with Path(path).open(encoding="utf-8") as stream: + document = yaml.load(stream, Loader=Yaml12SafeLoader) + except (OSError, yaml.YAMLError) as exc: + raise ConfigError(f"cannot load {path}: {exc}") from exc + if not isinstance(document, dict): + raise ConfigError(f"{path} must contain a YAML mapping") + return document + + +def apply_overrides( + document: dict[str, Any], assignments: list[str] +) -> dict[str, Any]: + """Replace existing values using JSON Pointer ``/path/to/value=YAML`` inputs.""" + + for assignment in assignments: + if "=" not in assignment: + raise ConfigError(f"override must have the form /path=value: {assignment}") + pointer, raw_value = assignment.split("=", 1) + if not pointer.startswith("/"): + raise ConfigError(f"override path must be a JSON Pointer: {pointer}") + tokens = [ + token.replace("~1", "/").replace("~0", "~") + for token in pointer.removeprefix("/").split("/") + ] + try: + value = yaml.load(raw_value, Loader=Yaml12SafeLoader) + except yaml.YAMLError as exc: + raise ConfigError(f"invalid YAML value for {pointer}: {exc}") from exc + + def list_index(token: str, length: int) -> int: + if re.fullmatch(r"0|[1-9][0-9]*", token) is None: + raise ConfigError(f"override path does not exist: {pointer}") + index = int(token) + if index >= length: + raise ConfigError(f"override path does not exist: {pointer}") + return index + + parent: Any = document + for token in tokens[:-1]: + if isinstance(parent, dict): + if token not in parent: + raise ConfigError(f"override path does not exist: {pointer}") + parent = parent[token] + elif isinstance(parent, list): + parent = parent[list_index(token, len(parent))] + else: + raise ConfigError(f"override path does not exist: {pointer}") + + leaf = tokens[-1] + if isinstance(parent, dict): + if leaf not in parent: + raise ConfigError(f"override path does not exist: {pointer}") + parent[leaf] = value + elif isinstance(parent, list): + parent[list_index(leaf, len(parent))] = value + else: + raise ConfigError(f"override path does not exist: {pointer}") + return document + + +def _placeholder_paths(value: Any, path: str = "") -> list[str]: + paths: list[str] = [] + if isinstance(value, dict): + for key, child in value.items(): + paths.extend(_placeholder_paths(child, f"{path}/{key}")) + elif isinstance(value, list): + for index, child in enumerate(value): + paths.extend(_placeholder_paths(child, f"{path}/{index}")) + elif isinstance(value, str) and re.search(r"<[^<>]+>", value): + paths.append(path or "/") + return paths + + +def render_document(document: dict[str, Any]) -> str: + """Serialize a concrete document with stable ordering and formatting.""" + + placeholders = _placeholder_paths(document) + if placeholders: + detail = ", ".join(placeholders) + raise ConfigError(f"unresolved angle-bracket placeholder(s): {detail}") + body = yaml.safe_dump( + document, + sort_keys=False, + default_flow_style=False, + allow_unicode=False, + width=1000, + ) + return "%YAML 1.2\n---\n" + body diff --git a/scripts/daqiri_config/profiles.py b/scripts/daqiri_config/profiles.py new file mode 100644 index 00000000..39134478 --- /dev/null +++ b/scripts/daqiri_config/profiles.py @@ -0,0 +1,749 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Reusable high-level DAQIRI configuration profiles.""" + +from __future__ import annotations + +import ipaddress +import re +from dataclasses import dataclass +from typing import Any + +from .core import ConfigError + + +SUPPORTED_TRANSFORMS = ("none", "vlan", "vxlan", "gre", "nvgre") +SUPPORTED_SOCKET_TRANSPORTS = ("udp", "tcp", "roce") +MAC_ADDRESS_RE = re.compile(r"^(?:[0-9a-fA-F]{2}:){5}[0-9a-fA-F]{2}$") +ETHERNET_HEADER_SIZE = 14 +UDP_IPV4_HEADER_SIZE = 42 +IPV4_MAX_TOTAL_LENGTH = 65535 +RAW_MAX_FRAME_SIZE = 9100 +RAW_TRANSFORM_WIRE_OVERHEAD = { + "none": 0, + "vlan": 4, + "vxlan": 14 + 20 + 8 + 8, + "gre": 14 + 20 + 4, + "nvgre": 14 + 20 + 8, +} + + +def _require_positive(name: str, value: int) -> None: + if value <= 0: + raise ConfigError(f"{name} must be greater than zero") + + +def _require_port(name: str, value: int) -> None: + if value <= 0 or value > 65535: + raise ConfigError(f"{name} must be in the range 1..65535") + + +def _require_ip(name: str, value: str) -> None: + try: + address = ipaddress.ip_address(value) + except ValueError as exc: + raise ConfigError(f"{name} must be an IPv4 address") from exc + if address.version != 4: + raise ConfigError(f"{name} must be an IPv4 address") + + +def _require_mac(name: str, value: str) -> None: + if MAC_ADDRESS_RE.fullmatch(value) is None: + raise ConfigError(f"{name} must be a six-octet MAC address") + + +def _endpoint(transport: str, address: str, port: int | None) -> str: + return f"{transport}://{address}" + (f":{port}" if port is not None else "") + + +@dataclass(frozen=True) +class SocketPairSpec: + """A unidirectional client/TX to server/RX socket or RoCE pair.""" + + transport: str + client_address: str + server_address: str + client_port: int + server_port: int + client_master_core: int + server_master_core: int + client_rx_core: int + client_tx_core: int + server_rx_core: int + server_tx_core: int + client_worker_core: int | None + server_worker_core: int | None + message_size: int + buffer_size: int + num_bufs: int + rx_num_bufs: int | None = None + tx_num_bufs: int | None = None + rx_batch_size: int | None = None + affinity: int = 0 + memory_kind: str | None = None + iterations: int | None = None + rx_depth: int | None = None + tx_depth: int | None = None + roce_transport_mode: str | None = None + include_benchmark: bool = True + + def __post_init__(self) -> None: + if self.transport not in SUPPORTED_SOCKET_TRANSPORTS: + raise ConfigError( + f"transport must be one of {', '.join(SUPPORTED_SOCKET_TRANSPORTS)}" + ) + if self.transport != "roce" and ( + self.rx_num_bufs is not None or self.tx_num_bufs is not None + ): + raise ConfigError("rx_num_bufs and tx_num_bufs are supported only for RoCE") + if self.transport == "roce" and self.rx_batch_size is not None: + raise ConfigError("rx_batch_size is supported only for TCP/UDP") + if self.transport == "roce" and self.iterations is not None: + raise ConfigError("iterations is supported only for TCP/UDP") + if self.transport != "roce" and ( + self.rx_depth is not None + or self.tx_depth is not None + or self.roce_transport_mode is not None + ): + raise ConfigError( + "rx_depth, tx_depth, and roce_transport_mode are supported only for RoCE" + ) + for name in ("client_port", "server_port"): + _require_port(name, getattr(self, name)) + for name in ("client_address", "server_address"): + _require_ip(name, getattr(self, name)) + for name in ( + "client_master_core", + "server_master_core", + "client_rx_core", + "client_tx_core", + "server_rx_core", + "server_tx_core", + ): + if getattr(self, name) < -1: + raise ConfigError(f"{name} must be -1 or a non-negative CPU index") + for name in ("client_worker_core", "server_worker_core"): + value = getattr(self, name) + if value is not None and value < -1: + raise ConfigError(f"{name} must be -1 or a non-negative CPU index") + if self.include_benchmark and ( + self.client_worker_core is None or self.server_worker_core is None + ): + raise ConfigError("benchmark profiles require client and server worker cores") + if self.affinity < 0: + raise ConfigError("affinity must not be negative") + for name in ( + "message_size", + "buffer_size", + "num_bufs", + ): + _require_positive(name, getattr(self, name)) + for name in ("rx_batch_size", "iterations", "rx_depth", "tx_depth"): + value = getattr(self, name) + if value is not None: + _require_positive(name, value) + if self.rx_num_bufs is not None: + _require_positive("rx_num_bufs", self.rx_num_bufs) + if self.tx_num_bufs is not None: + _require_positive("tx_num_bufs", self.tx_num_bufs) + if self.buffer_size < self.message_size: + raise ConfigError("buffer_size must be at least message_size") + if self.transport == "udp" and self.message_size > 65507: + raise ConfigError("UDP message_size must not exceed 65507 bytes") + rx_batch_size = self.rx_batch_size if self.rx_batch_size is not None else 1 + if self.transport == "udp" and rx_batch_size > 32: + raise ConfigError("UDP rx_batch_size must not exceed 32") + if self.transport in ("udp", "tcp") and rx_batch_size > self.num_bufs: + raise ConfigError("rx_batch_size must not exceed num_bufs") + if self.transport in ("udp", "tcp") and self.memory_kind == "device": + raise ConfigError("TCP/UDP socket profiles cannot use device memory") + if self.transport == "roce" and self.roce_transport_mode not in ( + None, + "RC", + "UC", + "UD", + ): + raise ConfigError("roce_transport_mode must be RC, UC, or UD") + if self.transport == "roce": + queue_cores = ( + self.client_master_core, + self.server_master_core, + self.client_rx_core, + self.client_tx_core, + self.server_rx_core, + self.server_tx_core, + ) + if any(core < 0 for core in queue_cores): + raise ConfigError("RoCE master and queue cores must be non-negative") + rx_depth = self.rx_depth if self.rx_depth is not None else 128 + tx_depth = self.tx_depth if self.tx_depth is not None else 128 + if rx_depth > (self.rx_num_bufs or self.num_bufs): + raise ConfigError("rx_depth must not exceed RX memory-region num_bufs") + if tx_depth > (self.tx_num_bufs or self.num_bufs): + raise ConfigError("tx_depth must not exceed TX memory-region num_bufs") + + +def _socket_memory_regions(spec: SocketPairSpec, role: str) -> list[dict[str, Any]]: + suffix = "CLIENT" if role == "tx" else "SERVER" + memory_kind = spec.memory_kind or ("host_pinned" if spec.transport == "roce" else "host") + if spec.transport == "roce": + return [ + { + "name": f"DATA_RX_GPU_{suffix}", + "kind": memory_kind, + "affinity": spec.affinity, + "num_bufs": spec.rx_num_bufs or spec.num_bufs, + "buf_size": spec.buffer_size, + }, + { + "name": f"DATA_TX_GPU_{suffix}", + "kind": memory_kind, + "affinity": spec.affinity, + "num_bufs": spec.tx_num_bufs or spec.num_bufs, + "buf_size": spec.buffer_size, + }, + ] + return [ + { + "name": f"DATA_SOCKET_{suffix}", + "kind": memory_kind, + "affinity": spec.affinity, + "num_bufs": spec.num_bufs, + "buf_size": spec.buffer_size, + } + ] + + +def _socket_role_document(spec: SocketPairSpec, role: str) -> dict[str, Any]: + is_client = role == "tx" + mode = "client" if is_client else "server" + suffix = mode.upper() + local_address = spec.client_address if is_client else spec.server_address + local_port = spec.client_port if is_client else spec.server_port + remote_address = spec.server_address if is_client else spec.client_address + remote_port = spec.server_port if is_client else spec.client_port + master_core = spec.client_master_core if is_client else spec.server_master_core + rx_core = spec.client_rx_core if is_client else spec.server_rx_core + tx_core = spec.client_tx_core if is_client else spec.server_tx_core + worker_core = spec.client_worker_core if is_client else spec.server_worker_core + memory_regions = _socket_memory_regions(spec, role) + + socket_config: dict[str, Any] = { + "mode": mode, + "local_addr": _endpoint( + spec.transport, + local_address, + None if spec.transport == "roce" and is_client else local_port, + ), + } + if spec.transport != "roce": + socket_config["remote_addr"] = _endpoint( + spec.transport, remote_address, remote_port + ) + socket_config["max_payload_size"] = min(65535, spec.buffer_size) + + interface: dict[str, Any] = { + "name": f"{spec.transport}_{mode}", + "address": local_address, + "socket_config": socket_config, + } + if spec.transport == "roce": + interface["roce_config"] = {"transport_mode": spec.roce_transport_mode or "RC"} + interface["rx"] = { + "queues": [ + { + "name": f"{suffix}_RX_Queue", + "id": 0, + "cpu_core": rx_core, + "batch_size": 1, + } + ] + } + interface["tx"] = { + "queues": [ + { + "name": f"{suffix}_TX_Queue", + "id": 0, + "cpu_core": tx_core, + "batch_size": 1, + } + ] + } + else: + region_name = memory_regions[0]["name"] + interface["rx"] = { + "queues": [ + { + "name": f"{suffix}_RX_Queue", + "id": 0, + "cpu_core": rx_core, + "batch_size": (spec.rx_batch_size or 1) if not is_client else 1, + "memory_regions": [region_name], + } + ] + } + interface["tx"] = { + "queues": [ + { + "name": f"{suffix}_TX_Queue", + "id": 0, + "cpu_core": tx_core, + "batch_size": 1, + "memory_regions": [region_name], + } + ] + } + + document: dict[str, Any] = { + "daqiri": { + "cfg": { + "version": 1, + "stream_type": "socket", + "master_core": master_core, + "debug": False, + "log_level": "info", + "memory_regions": memory_regions, + "interfaces": [interface], + } + } + } + + if spec.include_benchmark: + bench: dict[str, Any] = { + "cpu_core": worker_core, + "server": not is_client, + "send": is_client, + "receive": not is_client, + "message_size": spec.message_size, + "server_address": spec.server_address, + "server_port": spec.server_port, + } + if spec.transport == "roce": + bench["rx_depth"] = spec.rx_depth if spec.rx_depth is not None else 128 + bench["tx_depth"] = spec.tx_depth if spec.tx_depth is not None else 128 + if is_client: + bench["client_address"] = spec.client_address + document[f"rdma_bench_{mode}"] = bench + else: + bench["iterations"] = ( + spec.iterations if spec.iterations is not None else 1_000_000_000 + ) + if is_client: + bench["client_address"] = spec.client_address + document[f"socket_bench_{mode}"] = bench + + return document + + +def generate_socket_pair(spec: SocketPairSpec) -> dict[str, dict[str, Any]]: + """Generate concrete TX/client and RX/server documents.""" + + return { + "tx": _socket_role_document(spec, "tx"), + "rx": _socket_role_document(spec, "rx"), + } + + +@dataclass(frozen=True) +class RawPairSpec: + """A raw-Ethernet TX/RX pair, optionally multi-queue or transformed.""" + + tx_address: str + rx_address: str + master_core: int + tx_queue_cores: tuple[int, ...] + rx_queue_cores: tuple[int, ...] + tx_worker_cores: tuple[int, ...] + rx_worker_cores: tuple[int, ...] + eth_dst_addr: str + ip_src_addr: str = "1.1.1.1" + ip_dst_addr: str = "2.2.2.2" + udp_port_base: int = 4096 + payload_size: int = 8000 + header_size: int = 64 + buffer_size: int | None = None + batch_size: int | None = None + num_bufs: int = 51200 + affinity: int = 0 + memory_kind: str = "device" + engine: str | None = None + eth_src_addr: str | None = None + tx_eth_src: bool = True + transform: str = "none" + vlan_id: int = 100 + outer_eth_src: str = "02:00:00:00:00:01" + outer_eth_dst: str = "02:00:00:00:00:02" + outer_ipv4_src: str = "192.0.2.1" + outer_ipv4_dst: str = "192.0.2.2" + tunnel_id: int = 100 + include_benchmark: bool = True + + @property + def resolved_buffer_size(self) -> int: + """Return the configured packet-buffer capacity in bytes.""" + + if self.buffer_size is not None: + return self.buffer_size + return self.payload_size + self.header_size + + @property + def resolved_batch_size(self) -> int: + """Return the configured burst size or its backend-specific default.""" + + if self.batch_size is not None: + return self.batch_size + return 10240 if self.engine == "dpdk" else 1024 + + def __post_init__(self) -> None: + if not self.tx_queue_cores or not self.rx_queue_cores: + raise ConfigError("raw-pair requires at least one TX and one RX queue") + if self.include_benchmark and len(self.tx_worker_cores) != len(self.tx_queue_cores): + raise ConfigError("one tx_worker_core is required per TX queue") + if self.include_benchmark and len(self.rx_worker_cores) != len(self.rx_queue_cores): + raise ConfigError("one rx_worker_core is required per RX queue") + if not self.include_benchmark and self.tx_worker_cores and ( + len(self.tx_worker_cores) != len(self.tx_queue_cores) + ): + raise ConfigError("tx_worker_cores must be empty or match the TX queue count") + if not self.include_benchmark and self.rx_worker_cores and ( + len(self.rx_worker_cores) != len(self.rx_queue_cores) + ): + raise ConfigError("rx_worker_cores must be empty or match the RX queue count") + if len(self.tx_queue_cores) > 1 and len(self.rx_queue_cores) > 1: + if len(self.tx_queue_cores) != len(self.rx_queue_cores): + raise ConfigError( + "raw-pair supports unequal queue counts only when one side has one queue" + ) + if self.transform not in SUPPORTED_TRANSFORMS: + raise ConfigError( + f"transform must be one of {', '.join(SUPPORTED_TRANSFORMS)}" + ) + if self.transform != "none" and ( + len(self.tx_queue_cores) != 1 or len(self.rx_queue_cores) != 1 + ): + raise ConfigError( + "raw transform profiles require exactly one TX and one RX queue" + ) + if self.transform != "none" and self.engine is None: + raise ConfigError( + "raw transform profiles require an explicit dpdk or ibverbs engine" + ) + if self.engine not in (None, "dpdk", "ibverbs"): + raise ConfigError("raw-pair engine must be dpdk, ibverbs, or omitted") + if self.memory_kind not in ("huge", "device", "host_pinned", "host"): + raise ConfigError("unsupported memory_kind") + for name in ("payload_size", "header_size", "num_bufs"): + _require_positive(name, getattr(self, name)) + if self.batch_size is not None: + _require_positive("batch_size", self.batch_size) + batch_size = self.resolved_batch_size + _require_positive("buffer_size", self.resolved_buffer_size) + if self.num_bufs < batch_size: + raise ConfigError("num_bufs must be at least batch_size") + if self.engine in (None, "dpdk") and self.num_bufs < 2 * batch_size: + raise ConfigError("DPDK raw profiles require num_bufs to be at least twice batch_size") + if self.include_benchmark: + if self.eth_src_addr is None and self.engine in (None, "ibverbs"): + raise ConfigError( + "eth_src_addr is required for ibverbs or engine-default benchmark profiles" + ) + if self.header_size < UDP_IPV4_HEADER_SIZE: + raise ConfigError( + f"benchmark raw header_size must be at least {UDP_IPV4_HEADER_SIZE} bytes" + ) + if ( + self.header_size + self.payload_size - ETHERNET_HEADER_SIZE + > IPV4_MAX_TOTAL_LENGTH + ): + raise ConfigError( + "benchmark raw header_size + payload_size exceeds the IPv4 " + "total-length limit" + ) + if self.resolved_buffer_size < self.header_size + self.payload_size: + raise ConfigError( + "raw buffer_size must be at least header_size + payload_size " + "for benchmark profiles" + ) + wire_frame_size = ( + self.resolved_buffer_size + RAW_TRANSFORM_WIRE_OVERHEAD[self.transform] + ) + if wire_frame_size > RAW_MAX_FRAME_SIZE: + raise ConfigError( + "raw buffer_size + transform overhead " + f"({wire_frame_size} bytes) exceeds the runtime frame-size limit " + f"of {RAW_MAX_FRAME_SIZE} bytes" + ) + for name in ("master_core", "tx_queue_cores", "rx_queue_cores"): + values = getattr(self, name) + values = values if isinstance(values, tuple) else (values,) + if any(value < 0 for value in values): + raise ConfigError(f"{name} must contain non-negative CPU indices") + for name in ("tx_worker_cores", "rx_worker_cores"): + if any(value < -1 for value in getattr(self, name)): + raise ConfigError(f"{name} must contain -1 or non-negative CPU indices") + if self.affinity < 0: + raise ConfigError("affinity must not be negative") + _require_mac("eth_dst_addr", self.eth_dst_addr) + if self.eth_src_addr is not None: + _require_mac("eth_src_addr", self.eth_src_addr) + _require_ip("ip_src_addr", self.ip_src_addr) + _require_ip("ip_dst_addr", self.ip_dst_addr) + if self.transform in ("vxlan", "gre", "nvgre"): + _require_mac("outer_eth_src", self.outer_eth_src) + _require_mac("outer_eth_dst", self.outer_eth_dst) + _require_ip("outer_ipv4_src", self.outer_ipv4_src) + _require_ip("outer_ipv4_dst", self.outer_ipv4_dst) + if not 0 <= self.vlan_id <= 4095: + raise ConfigError("vlan_id must be in the range 0..4095") + if not 0 <= self.tunnel_id <= 0xFFFFFF: + raise ConfigError("tunnel_id must be in the range 0..16777215") + _require_port("udp_port_base", self.udp_port_base) + flow_count = max(len(self.tx_queue_cores), len(self.rx_queue_cores)) + if self.udp_port_base + flow_count - 1 > 65535: + raise ConfigError("raw-pair UDP port range exceeds 65535") + + +def _raw_action(transform: str, direction: str, spec: RawPairSpec) -> dict[str, Any]: + if transform == "vlan": + if direction == "tx": + return { + "type": "vlan_push", + "vlan_id": spec.vlan_id, + "pcp": 0, + "dei": 0, + "ethertype": 0x8100, + } + return {"type": "vlan_pop"} + + tunnel: dict[str, Any] = { + "type": transform, + "outer_eth_src": spec.outer_eth_src, + "outer_eth_dst": spec.outer_eth_dst, + "outer_ipv4_src": spec.outer_ipv4_src, + "outer_ipv4_dst": spec.outer_ipv4_dst, + } + if transform == "vxlan": + tunnel.update( + { + "outer_udp_src": 49152, + "outer_udp_dst": 4789, + "vni": spec.tunnel_id, + } + ) + elif transform == "gre": + tunnel["gre_protocol"] = 0x0800 + elif transform == "nvgre": + tunnel.update({"tni": spec.tunnel_id, "flow_id": 0}) + return { + "type": "tunnel_encap" if direction == "tx" else "tunnel_decap", + "tunnel": tunnel, + } + + +def _raw_transform_rx_match(spec: RawPairSpec) -> dict[str, Any]: + """Return the backend-appropriate match surrounding an RX transform.""" + + # DPDK derives the outer tunnel/VLAN pattern from the decap action and + # treats the explicit match as an optional inner-packet match. Keep the + # map present so the C++ YAML decoder does not discard the RX section. + if spec.engine == "dpdk": + return {} + if spec.transform == "vlan": + return {"udp_src": spec.udp_port_base, "udp_dst": spec.udp_port_base} + if spec.transform == "vxlan": + return {"udp_src": 49152, "udp_dst": 4789} + return { + "ipv4_src": spec.outer_ipv4_src, + "ipv4_dst": spec.outer_ipv4_dst, + } + + +def _raw_regions(prefix: str, count: int, spec: RawPairSpec) -> list[dict[str, Any]]: + numbered = count > 1 + return [ + { + "name": f"Data_{prefix}_GPU" + (f"_{index}" if numbered else ""), + "kind": spec.memory_kind, + "affinity": spec.affinity, + "num_bufs": spec.num_bufs, + "buf_size": spec.resolved_buffer_size, + } + for index in range(count) + ] + + +def generate_raw_pair(spec: RawPairSpec) -> dict[str, Any]: + """Generate one concrete raw-Ethernet benchmark or production document.""" + + tx_count = len(spec.tx_queue_cores) + rx_count = len(spec.rx_queue_cores) + flow_count = max(tx_count, rx_count) + tx_regions = _raw_regions("TX", tx_count, spec) + rx_regions = _raw_regions("RX", rx_count, spec) + + tx_queues = [] + for index, core in enumerate(spec.tx_queue_cores): + queue = { + "name": f"tx_q_{index}", + "id": index, + "batch_size": spec.resolved_batch_size, + "cpu_core": core, + "memory_regions": [tx_regions[index]["name"]], + } + if spec.tx_eth_src: + queue["offloads"] = ["tx_eth_src"] + tx_queues.append(queue) + + rx_queues = [] + for index, core in enumerate(spec.rx_queue_cores): + rx_queues.append( + { + "name": f"rx_q_{index}", + "id": index, + "cpu_core": core, + "batch_size": spec.resolved_batch_size, + "memory_regions": [rx_regions[index]["name"]], + } + ) + + rx_flows = [] + for index in range(flow_count): + flow: dict[str, Any] = { + "name": f"flow_{index}", + # The ibverbs raw engine reserves zero as an invalid flow ID. Keep + # generated profiles portable across both raw engines. + "id": index + 1, + "action": {"type": "queue", "id": index % rx_count}, + "match": { + "udp_src": spec.udp_port_base + index, + "udp_dst": spec.udp_port_base + index, + }, + } + if spec.transform != "none": + flow["name"] = f"{spec.transform}_decap" + flow["id"] = 100 + SUPPORTED_TRANSFORMS.index(spec.transform) + flow.pop("action") + flow["match"] = _raw_transform_rx_match(spec) + flow["actions"] = [ + _raw_action(spec.transform, "rx", spec), + {"type": "queue", "id": index % rx_count}, + ] + rx_flows.append(flow) + + tx: dict[str, Any] = {"queues": tx_queues} + if spec.transform != "none": + tx["flows"] = [ + { + "name": f"{spec.transform}_encap", + "id": 100 + SUPPORTED_TRANSFORMS.index(spec.transform), + "actions": [_raw_action(spec.transform, "tx", spec)], + "match": { + "udp_src": spec.udp_port_base, + "udp_dst": spec.udp_port_base, + }, + } + ] + + config: dict[str, Any] = { + "version": 1, + "stream_type": "raw", + } + if spec.engine is not None: + config["engine"] = spec.engine + config.update({ + "master_core": spec.master_core, + "debug": False, + "log_level": "info", + "loopback": "", + "memory_regions": [*tx_regions, *rx_regions], + "interfaces": [ + {"name": "tx_port", "address": spec.tx_address, "tx": tx}, + { + "name": "rx_port", + "address": spec.rx_address, + "rx": { + "flow_isolation": True, + "queues": rx_queues, + "flows": rx_flows, + }, + }, + ], + }) + + document: dict[str, Any] = {"daqiri": {"cfg": config}} + if spec.include_benchmark: + document["bench_rx"] = [ + { + "interface_name": "rx_port", + "queue_id": index, + "cpu_core": spec.rx_worker_cores[index], + } + for index in range(rx_count) + ] + bench_tx = [] + for index in range(tx_count): + first_port = spec.udp_port_base + index + if tx_count == 1 and flow_count > 1: + port: int | str = f"{spec.udp_port_base}-{spec.udp_port_base + flow_count - 1}" + else: + port = first_port + entry = { + "interface_name": "tx_port", + "queue_id": index, + "cpu_core": spec.tx_worker_cores[index], + "batch_size": spec.resolved_batch_size, + "payload_size": spec.payload_size, + "header_size": spec.header_size, + "eth_dst_addr": spec.eth_dst_addr, + "ip_src_addr": spec.ip_src_addr, + "ip_dst_addr": spec.ip_dst_addr, + "udp_src_port": port, + "udp_dst_port": port, + } + if spec.eth_src_addr is not None: + entry["eth_src_addr"] = spec.eth_src_addr + bench_tx.append(entry) + document["bench_tx"] = bench_tx + + return document + + +def generate_raw_roles(spec: RawPairSpec) -> dict[str, dict[str, Any]]: + """Split a raw pair into independently runnable TX-only and RX-only documents.""" + + combined = generate_raw_pair(spec) + config = combined["daqiri"]["cfg"] + tx_interface, rx_interface = config["interfaces"] + tx_region_names = { + name + for queue in tx_interface["tx"]["queues"] + for name in queue.get("memory_regions", []) + } + rx_region_names = { + name + for queue in rx_interface["rx"]["queues"] + for name in queue.get("memory_regions", []) + } + + def role_document( + interface: dict[str, Any], region_names: set[str], bench_key: str + ) -> dict[str, Any]: + role_config = { + key: value + for key, value in config.items() + if key not in ("memory_regions", "interfaces") + } + role_config["memory_regions"] = [ + region for region in config["memory_regions"] if region["name"] in region_names + ] + role_config["interfaces"] = [interface] + document: dict[str, Any] = {"daqiri": {"cfg": role_config}} + if spec.include_benchmark: + document[bench_key] = combined[bench_key] + return document + + return { + "tx": role_document(tx_interface, tx_region_names, "bench_tx"), + "rx": role_document(rx_interface, rx_region_names, "bench_rx"), + } diff --git a/scripts/gen_daqiri_config.py b/scripts/gen_daqiri_config.py new file mode 100755 index 00000000..27f26bb5 --- /dev/null +++ b/scripts/gen_daqiri_config.py @@ -0,0 +1,329 @@ +#!/usr/bin/env python3 +# +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Generate deterministic production and benchmark DAQIRI configurations.""" + +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +# CMake replaces this token in the installed launcher with either a path +# relative to CMAKE_INSTALL_BINDIR or an absolute GNUInstallDirs data path. +# In the source tree the unresolved token is ignored and Python finds the +# sibling daqiri_config package normally. +CONFIGURED_MODULE_HINT = "@DAQIRI_CONFIG_GENERATOR_MODULE_HINT@" +UNCONFIGURED_MODULE_HINT = "@" + "DAQIRI_CONFIG_GENERATOR_MODULE_HINT" + "@" +if CONFIGURED_MODULE_HINT != UNCONFIGURED_MODULE_HINT: + installed_modules = Path(CONFIGURED_MODULE_HINT) + if not installed_modules.is_absolute(): + installed_modules = Path(__file__).resolve().parent / installed_modules + sys.path.insert(0, str(installed_modules.resolve())) + +from daqiri_config import ( + ConfigError, + RawPairSpec, + SocketPairSpec, + apply_overrides, + generate_raw_pair, + generate_raw_roles, + generate_socket_pair, + load_document, + render_document, +) + + +def _cores(value: str) -> tuple[int, ...]: + try: + cores = tuple(int(item) for item in value.split(",")) + except ValueError as exc: + raise argparse.ArgumentTypeError("expected a comma-separated CPU list") from exc + if not cores: + raise argparse.ArgumentTypeError("CPU list must not be empty") + return cores + + +def _add_output_argument(parser: argparse.ArgumentParser) -> None: + parser.add_argument("-o", "--output", help="write one generated document to this path") + + +def _write(text: str, output: str | None) -> None: + if output: + Path(output).write_text(text, encoding="utf-8") + else: + sys.stdout.write(text) + + +def _socket_parser(subparsers: argparse._SubParsersAction) -> None: + parser = subparsers.add_parser( + "socket-pair", + help="generate a unidirectional TCP, UDP, or RoCE TX/RX pair", + ) + parser.add_argument("--transport", choices=("udp", "tcp", "roce"), required=True) + parser.add_argument("--client-address", required=True, help="TX/client IP address") + parser.add_argument("--server-address", required=True, help="RX/server IP address") + parser.add_argument("--client-port", type=int, required=True) + parser.add_argument("--server-port", type=int, required=True) + parser.add_argument("--client-master-core", type=int, required=True) + parser.add_argument("--server-master-core", type=int, required=True) + parser.add_argument("--client-rx-core", type=int, required=True) + parser.add_argument("--client-tx-core", type=int, required=True) + parser.add_argument("--server-rx-core", type=int, required=True) + parser.add_argument("--server-tx-core", type=int, required=True) + parser.add_argument("--client-worker-core", type=int) + parser.add_argument("--server-worker-core", type=int) + parser.add_argument("--message-size", type=int, required=True) + parser.add_argument("--buffer-size", type=int, required=True) + parser.add_argument("--num-bufs", type=int, required=True) + parser.add_argument( + "--rx-num-bufs", type=int, help="RoCE RX memory-region buffer count" + ) + parser.add_argument( + "--tx-num-bufs", type=int, help="RoCE TX memory-region buffer count" + ) + parser.add_argument( + "--rx-batch-size", type=int, help="TCP/UDP RX batch size (default: 1)" + ) + parser.add_argument("--affinity", type=int, default=0) + parser.add_argument("--memory-kind", choices=("huge", "device", "host_pinned", "host")) + parser.add_argument( + "--iterations", type=int, help="TCP/UDP benchmark iterations (default: 1000000000)" + ) + parser.add_argument("--rx-depth", type=int, help="RoCE RX depth (default: 128)") + parser.add_argument("--tx-depth", type=int, help="RoCE TX depth (default: 128)") + parser.add_argument( + "--roce-transport-mode", + choices=("RC", "UC", "UD"), + help="RoCE transport mode (default: RC)", + ) + parser.add_argument( + "--role", + choices=("tx", "rx", "both"), + default="both", + help="emit one role to stdout/output, or both roles to --output-dir", + ) + parser.add_argument("--output-dir", help="directory for tx.yaml and rx.yaml") + parser.add_argument( + "--daqiri-only", + action="store_true", + help="omit benchmark-owned top-level sections", + ) + _add_output_argument(parser) + + +def _raw_parser(subparsers: argparse._SubParsersAction) -> None: + parser = subparsers.add_parser( + "raw-pair", help="generate a raw-Ethernet TX/RX configuration" + ) + parser.add_argument("--tx-address", required=True, help="TX PCI BDF or interface") + parser.add_argument("--rx-address", required=True, help="RX PCI BDF or interface") + parser.add_argument("--master-core", type=int, required=True) + parser.add_argument("--tx-queue-cores", type=_cores, required=True) + parser.add_argument("--rx-queue-cores", type=_cores, required=True) + parser.add_argument("--tx-worker-cores", type=_cores, default=()) + parser.add_argument("--rx-worker-cores", type=_cores, default=()) + parser.add_argument("--eth-dst-addr", required=True) + parser.add_argument("--ip-src-addr", default="1.1.1.1") + parser.add_argument("--ip-dst-addr", default="2.2.2.2") + parser.add_argument("--udp-port-base", type=int, default=4096) + parser.add_argument("--payload-size", type=int, default=8000) + parser.add_argument("--header-size", type=int, default=64) + parser.add_argument( + "--buffer-size", + "--buf-size", + type=int, + help="packet-buffer capacity (default: header size + payload size)", + ) + parser.add_argument( + "--batch-size", + type=int, + help="packets per burst (default: 10240 for dpdk, 1024 for ibverbs or engine default)", + ) + parser.add_argument("--num-bufs", type=int, default=51200) + parser.add_argument("--affinity", type=int, default=0) + parser.add_argument( + "--memory-kind", + choices=("huge", "device", "host_pinned", "host"), + default="device", + ) + parser.add_argument("--engine", choices=("dpdk", "ibverbs")) + parser.add_argument( + "--eth-src-addr", + help="Ethernet source address for benchmark packet templates; required for ibverbs benchmark profiles", + ) + parser.add_argument( + "--tx-eth-src", + action=argparse.BooleanOptionalAction, + default=True, + help="enable NIC insertion of the Ethernet source address (default: enabled)", + ) + parser.add_argument( + "--transform", choices=("none", "vlan", "vxlan", "gre", "nvgre"), default="none" + ) + parser.add_argument("--vlan-id", type=int, default=100) + parser.add_argument("--outer-eth-src", default="02:00:00:00:00:01") + parser.add_argument("--outer-eth-dst", default="02:00:00:00:00:02") + parser.add_argument("--outer-ipv4-src", default="192.0.2.1") + parser.add_argument("--outer-ipv4-dst", default="192.0.2.2") + parser.add_argument("--tunnel-id", type=int, default=100) + parser.add_argument( + "--role", + choices=("loopback", "tx", "rx", "both"), + default="loopback", + help="emit a combined loopback document, one role, or a TX/RX pair", + ) + parser.add_argument("--output-dir", help="directory for tx.yaml and rx.yaml") + parser.add_argument( + "--daqiri-only", + action="store_true", + help="omit benchmark-owned top-level sections", + ) + _add_output_argument(parser) + + +def _render_parser(subparsers: argparse._SubParsersAction) -> None: + parser = subparsers.add_parser( + "render", help="deterministically render an arbitrary DAQIRI document" + ) + parser.add_argument("input", help="YAML document or bare daqiri.cfg mapping") + parser.add_argument( + "--set", + dest="overrides", + action="append", + default=[], + metavar="/JSON/POINTER=VALUE", + help="replace an existing value before rendering; repeat as needed", + ) + _add_output_argument(parser) + + +def _socket_spec(args: argparse.Namespace) -> SocketPairSpec: + return SocketPairSpec( + transport=args.transport, + client_address=args.client_address, + server_address=args.server_address, + client_port=args.client_port, + server_port=args.server_port, + client_master_core=args.client_master_core, + server_master_core=args.server_master_core, + client_rx_core=args.client_rx_core, + client_tx_core=args.client_tx_core, + server_rx_core=args.server_rx_core, + server_tx_core=args.server_tx_core, + client_worker_core=args.client_worker_core, + server_worker_core=args.server_worker_core, + message_size=args.message_size, + buffer_size=args.buffer_size, + num_bufs=args.num_bufs, + rx_num_bufs=args.rx_num_bufs, + tx_num_bufs=args.tx_num_bufs, + rx_batch_size=args.rx_batch_size, + affinity=args.affinity, + memory_kind=args.memory_kind, + iterations=args.iterations, + rx_depth=args.rx_depth, + tx_depth=args.tx_depth, + roce_transport_mode=args.roce_transport_mode, + include_benchmark=not args.daqiri_only, + ) + + +def _raw_spec(args: argparse.Namespace) -> RawPairSpec: + return RawPairSpec( + tx_address=args.tx_address, + rx_address=args.rx_address, + master_core=args.master_core, + tx_queue_cores=args.tx_queue_cores, + rx_queue_cores=args.rx_queue_cores, + tx_worker_cores=args.tx_worker_cores, + rx_worker_cores=args.rx_worker_cores, + eth_dst_addr=args.eth_dst_addr, + ip_src_addr=args.ip_src_addr, + ip_dst_addr=args.ip_dst_addr, + udp_port_base=args.udp_port_base, + payload_size=args.payload_size, + header_size=args.header_size, + buffer_size=args.buffer_size, + batch_size=args.batch_size, + num_bufs=args.num_bufs, + affinity=args.affinity, + memory_kind=args.memory_kind, + engine=args.engine, + eth_src_addr=args.eth_src_addr, + tx_eth_src=args.tx_eth_src, + transform=args.transform, + vlan_id=args.vlan_id, + outer_eth_src=args.outer_eth_src, + outer_eth_dst=args.outer_eth_dst, + outer_ipv4_src=args.outer_ipv4_src, + outer_ipv4_dst=args.outer_ipv4_dst, + tunnel_id=args.tunnel_id, + include_benchmark=not args.daqiri_only, + ) + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__) + subparsers = parser.add_subparsers(dest="command", required=True) + _render_parser(subparsers) + _socket_parser(subparsers) + _raw_parser(subparsers) + args = parser.parse_args(argv) + + try: + if args.command == "render": + document = apply_overrides(load_document(args.input), args.overrides) + _write(render_document(document), args.output) + return 0 + + if args.command == "raw-pair": + spec = _raw_spec(args) + if args.role == "loopback": + if args.output_dir: + parser.error("raw-pair --role loopback does not use --output-dir") + _write(render_document(generate_raw_pair(spec)), args.output) + return 0 + documents = generate_raw_roles(spec) + if args.role == "both": + if args.output or not args.output_dir: + parser.error( + "raw-pair --role both requires --output-dir and does not use --output" + ) + output_dir = Path(args.output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + for role, document in documents.items(): + (output_dir / f"{role}.yaml").write_text( + render_document(document), encoding="utf-8" + ) + else: + if args.output_dir: + parser.error("--output-dir is only valid with --role both") + _write(render_document(documents[args.role]), args.output) + return 0 + + documents = generate_socket_pair(_socket_spec(args)) + if args.role == "both": + if args.output or not args.output_dir: + parser.error( + "socket-pair --role both requires --output-dir and does not use --output" + ) + output_dir = Path(args.output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + for role, document in documents.items(): + (output_dir / f"{role}.yaml").write_text( + render_document(document), encoding="utf-8" + ) + else: + if args.output_dir: + parser.error("--output-dir is only valid with --role both") + _write(render_document(documents[args.role]), args.output) + return 0 + except ConfigError as exc: + parser.exit(2, f"error: {exc}\n") + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/gen_spark_mq_config.py b/scripts/gen_spark_mq_config.py deleted file mode 100755 index 9b749cbc..00000000 --- a/scripts/gen_spark_mq_config.py +++ /dev/null @@ -1,106 +0,0 @@ -#!/usr/bin/env python3 -# -# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -# -"""Derive a DGX Spark multi-queue core-scaling cell from the (TX=2, RX=2) base. - -run_spark_mq_bench.sh sweeps the matrix cells (TX,RX) = (1,1),(1,2),(2,1),(2,2). -Rather than check in four near-identical YAMLs, we keep one superset base -- -examples/daqiri_bench_raw_tx_rx_spark_mq.yaml -- and prune it down to each cell -here. The pruning mirrors exactly what the four hand-written configs used to be: - - * memory regions / TX queues / RX queues / bench entries: keep index 0, and - keep index 1 only when that side has 2 queues. - * flows: keep flow_0; keep flow_1 when there are two distinct UDP ports on the - wire, i.e. when EITHER side has 2 queues (max(tx, rx) == 2). flow_1 routes to - RX queue 1 when RX has 2 queues, else to RX queue 0 (the 2t1r case, where two - TX flows both land on the lone RX queue). - * bench_tx UDP ports: a single TX worker feeding two RX queues (1t2r) round- - robins the port range "4096-4097"; otherwise queue N uses port 4096+N. - -PyYAML round-trips structure (not comments); the bench's yaml-cpp parser only -reads parsed content, so dropping comments/key-order is harmless. Output goes to -stdout. -""" - -from __future__ import annotations - -import argparse -import sys - -import yaml - -TX_PORT_BASE = 4096 - - -def prune(base: dict, tx: int, rx: int, payload: int, batch: int | None, - eth_dst: str | None) -> dict: - cfg = base["daqiri"]["cfg"] - n_ports = max(tx, rx) - - # Memory regions: keep _GPU_0 always, _GPU_1 per that side's count. - def keep_region(name: str) -> bool: - if name.endswith("_1"): - return tx == 2 if "_TX_" in name else rx == 2 - return True - - cfg["memory_regions"] = [m for m in cfg["memory_regions"] - if keep_region(m["name"])] - - for iface in cfg["interfaces"]: - if "tx" in iface: - iface["tx"]["queues"] = [q for q in iface["tx"]["queues"] - if q["id"] == 0 or tx == 2] - if "rx" in iface: - iface["rx"]["queues"] = [q for q in iface["rx"]["queues"] - if q["id"] == 0 or rx == 2] - flows = [] - for f in iface["rx"]["flows"]: - if f["id"] == 0: - flows.append(f) - elif n_ports == 2: - f["action"]["id"] = 1 if rx == 2 else 0 - flows.append(f) - iface["rx"]["flows"] = flows - - base["bench_rx"] = [b for b in base["bench_rx"] - if b["queue_id"] == 0 or rx == 2] - - bench_tx = [b for b in base["bench_tx"] if b["queue_id"] == 0 or tx == 2] - for b in bench_tx: - b["payload_size"] = payload - if batch is not None: - b["batch_size"] = batch - if eth_dst is not None: - b["eth_dst_addr"] = eth_dst - if b["queue_id"] == 0 and tx == 1 and rx == 2: - # Single TX worker round-robins both flows' ports. - b["udp_src_port"] = f"{TX_PORT_BASE}-{TX_PORT_BASE + 1}" - b["udp_dst_port"] = f"{TX_PORT_BASE}-{TX_PORT_BASE + 1}" - base["bench_tx"] = bench_tx - - return base - - -def main() -> int: - ap = argparse.ArgumentParser(description=__doc__) - ap.add_argument("base", help="path to the (2,2) superset base YAML") - ap.add_argument("--tx", type=int, choices=(1, 2), required=True) - ap.add_argument("--rx", type=int, choices=(1, 2), required=True) - ap.add_argument("--payload", type=int, required=True) - ap.add_argument("--batch", type=int, default=None) - ap.add_argument("--eth-dst", default=None, - help="fill bench_tx eth_dst_addr (the rx_port MAC)") - args = ap.parse_args() - - with open(args.base, encoding="utf-8") as fh: - base = yaml.safe_load(fh) - - out = prune(base, args.tx, args.rx, args.payload, args.batch, args.eth_dst) - yaml.safe_dump(out, sys.stdout, sort_keys=False, default_flow_style=False) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/scripts/gen_spark_netns_config.py b/scripts/gen_spark_netns_config.py deleted file mode 100755 index 2284da15..00000000 --- a/scripts/gen_spark_netns_config.py +++ /dev/null @@ -1,105 +0,0 @@ -#!/usr/bin/env python3 -# -# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -# -"""Split a combined DGX Spark netns bench base into a single-role config. - -The socket/RDMA wire-loopback sweep runs the server and client in separate -network namespaces, so each process needs a config that names only its own -interface (a combined two-interface config would make a process init the other -namespace's IP, which it does not own). Rather than check in a server YAML and a -client YAML per protocol, we keep one combined base -- both interfaces, both -memory regions, both bench sections -- and split it to a role here: - - * interfaces: keep the one whose socket_config.mode == role. - * memory regions: keep those whose name contains the role (SERVER / CLIENT). - * bench sections: drop the other role's _bench_ mapping. - * optional queue overrides: update RX batch size plus RX queue, TX queue, and - benchmark-worker affinity without text substitutions that conflate them. - -This is the structural inverse of unioning the old _netns_server / _netns_client -files. run_spark_bench.sh supplies optional structured queue overrides, then -pipes the output through per-message-size awk/sed rewrites -(num_bufs/buf_size/depths for RDMA; payload and ports for sockets). Output goes -to stdout. -""" - -from __future__ import annotations - -import argparse -import sys - -import yaml - - -def split_role( - base: dict, - role: str, - rx_queue_cpu_core: int | None = None, - rx_queue_batch_size: int | None = None, - tx_queue_cpu_core: int | None = None, - bench_cpu_core: int | None = None, -) -> dict: - cfg = base["daqiri"]["cfg"] - cfg["interfaces"] = [ - i for i in cfg["interfaces"] - if i.get("socket_config", {}).get("mode") == role - ] - if not cfg["interfaces"]: - raise SystemExit(f"no interface with socket_config.mode == {role!r}") - if rx_queue_cpu_core is not None: - for interface in cfg["interfaces"]: - for queue in interface.get("rx", {}).get("queues", []): - queue["cpu_core"] = rx_queue_cpu_core - if rx_queue_batch_size is not None: - for interface in cfg["interfaces"]: - for queue in interface.get("rx", {}).get("queues", []): - queue["batch_size"] = rx_queue_batch_size - if tx_queue_cpu_core is not None: - for interface in cfg["interfaces"]: - for queue in interface.get("tx", {}).get("queues", []): - queue["cpu_core"] = tx_queue_cpu_core - cfg["memory_regions"] = [ - m for m in cfg["memory_regions"] if role.upper() in m["name"] - ] - - other = "client" if role == "server" else "server" - for key in [k for k in base if k.endswith(f"_bench_{other}")]: - del base[key] - role_bench_keys = [k for k in base if k.endswith(f"_bench_{role}")] - if not role_bench_keys: - raise SystemExit(f"base has no *_bench_{role} section") - if bench_cpu_core is not None: - for key in role_bench_keys: - base[key]["cpu_core"] = bench_cpu_core - return base - - -def main() -> int: - ap = argparse.ArgumentParser(description=__doc__) - ap.add_argument("base", help="path to the combined (both-role) netns base YAML") - ap.add_argument("--role", choices=("server", "client"), required=True) - ap.add_argument("--rx-queue-cpu-core", type=int) - ap.add_argument("--rx-queue-batch-size", type=int) - ap.add_argument("--tx-queue-cpu-core", type=int) - ap.add_argument("--bench-cpu-core", type=int) - args = ap.parse_args() - - with open(args.base, encoding="utf-8") as fh: - base = yaml.safe_load(fh) - - out = split_role( - base, - args.role, - rx_queue_cpu_core=args.rx_queue_cpu_core, - rx_queue_batch_size=args.rx_queue_batch_size, - tx_queue_cpu_core=args.tx_queue_cpu_core, - bench_cpu_core=args.bench_cpu_core, - ) - yaml.safe_dump(out, sys.stdout, sort_keys=False, default_flow_style=False) - return 0 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/scripts/setup_spark_rdma_loopback.sh b/scripts/setup_spark_rdma_loopback.sh index 353a4954..486f1a3e 100755 --- a/scripts/setup_spark_rdma_loopback.sh +++ b/scripts/setup_spark_rdma_loopback.sh @@ -3,7 +3,7 @@ # Adapted from a colleague's single-adapter script for this host's # inter-port loopback: Adapter1:port0 (1.1.1.1) <-> Adapter2:port1 (2.2.2.2). # -# Matches examples/daqiri_bench_rdma_tx_rx_spark.yaml (1.1.1.1 / 2.2.2.2). +# Matches the generated Spark RoCE pair (1.1.1.1 / 2.2.2.2). # Re-running is safe: replaces addresses, flushes per-port tables, and # deletes any matching rules before re-adding. # diff --git a/scripts/setup_spark_xhost_net.sh b/scripts/setup_spark_xhost_net.sh index e77d6423..b5efb4d1 100755 --- a/scripts/setup_spark_xhost_net.sh +++ b/scripts/setup_spark_xhost_net.sh @@ -3,9 +3,9 @@ # Adds the host route the kernel needs to reach the peer # when daqiri-tx / daqiri-rx nmcli profiles are split across two boxes. # -# Matches examples/*_spark_xhost.yaml (1.1.1.1 on TX, 2.2.2.2 on RX). -# Re-running is safe: replaces the route. An optional peer MAC preserves the -# old static-neighbor setup, but DAQIRI's ibverbs raw engine can resolve it. +# Matches raw-pair configs generated with 1.1.1.1 on TX and 2.2.2.2 on RX. +# Re-running replaces the route. --peer-mac optionally installs a static +# neighbor; without it, the kernel can resolve the peer with ARP. # # Conflicts with scripts/setup_spark_rdma_loopback.sh on the same host: that # script reassigns 1.1.1.1 / 2.2.2.2 across two local ports for inter-port diff --git a/tests/portable/test_config_generator_core.py b/tests/portable/test_config_generator_core.py new file mode 100644 index 00000000..a451f652 --- /dev/null +++ b/tests/portable/test_config_generator_core.py @@ -0,0 +1,254 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import shutil +import subprocess +import sys +from pathlib import Path + +import pytest +import yaml + + +REPOSITORY_ROOT = Path(__file__).resolve().parents[2] + +from scripts.daqiri_config import ( + ConfigError, + apply_overrides, + render_document, +) + + +def minimal_document() -> dict: + return { + "daqiri": { + "cfg": { + "version": 1, + "stream_type": "raw", + "master_core": 1, + "memory_regions": [ + { + "name": "RX", + "kind": "host", + "affinity": 0, + "num_bufs": 8, + "buf_size": 2048, + } + ], + "interfaces": [ + { + "name": "rx", + "address": "0000:01:00.1", + "rx": { + "queues": [ + { + "name": "rxq0", + "id": 0, + "cpu_core": 2, + "batch_size": 4, + "memory_regions": ["RX"], + } + ] + }, + } + ], + } + }, + "application_owned": {"arbitrary": True}, + } + + +def test_overrides_fill_typed_placeholders() -> None: + document = minimal_document() + document["daqiri"]["cfg"]["master_core"] = "" + apply_overrides(document, ["/daqiri/cfg/master_core=3"]) + assert document["daqiri"]["cfg"]["master_core"] == 3 + + +def test_override_rejects_unknown_path() -> None: + with pytest.raises(ConfigError, match="does not exist"): + apply_overrides(minimal_document(), ["/daqiri/cfg/master_cores=3"]) + + +@pytest.mark.parametrize("index", ["-1", "01", "+1", "2"]) +def test_override_rejects_invalid_array_index(index: str) -> None: + with pytest.raises(ConfigError, match="does not exist"): + apply_overrides( + minimal_document(), + [f"/daqiri/cfg/interfaces/{index}/name=changed"], + ) + + +def test_render_rejects_application_owned_placeholders() -> None: + document = minimal_document() + document["application_owned"]["address"] = "" + with pytest.raises(ConfigError, match="/application_owned/address"): + render_document(document) + + +def test_render_is_byte_deterministic_and_quotes_pci_bdf() -> None: + first = render_document(minimal_document()) + second = render_document(minimal_document()) + assert first == second + assert first.startswith("%YAML 1.2\n---\n") + assert "address: '0000:01:00.1'" in first + assert yaml.safe_load(first)["application_owned"] == {"arbitrary": True} + + +def test_render_cli_accepts_bare_network_mapping(tmp_path: Path) -> None: + bare = minimal_document()["daqiri"]["cfg"] + source = tmp_path / "input.yaml" + source.write_text(yaml.safe_dump(bare, sort_keys=False), encoding="utf-8") + result = subprocess.run( + [ + sys.executable, + str(REPOSITORY_ROOT / "scripts/gen_daqiri_config.py"), + "render", + str(source), + ], + check=True, + capture_output=True, + text=True, + ) + assert yaml.safe_load(result.stdout) == bare + + +def test_installed_launcher_supports_custom_gnu_data_directory(tmp_path: Path) -> None: + prefix = tmp_path / "prefix" + bindir = prefix / "tools" + module_root = prefix / "libdata" / "daqiri" / "config-generator" + bindir.mkdir(parents=True) + shutil.copytree( + REPOSITORY_ROOT / "scripts/daqiri_config", + module_root / "daqiri_config", + ignore=shutil.ignore_patterns("__pycache__"), + ) + + launcher = (REPOSITORY_ROOT / "scripts/gen_daqiri_config.py").read_text( + encoding="utf-8" + ) + launcher = launcher.replace( + "@DAQIRI_CONFIG_GENERATOR_MODULE_HINT@", + "../libdata/daqiri/config-generator", + ) + launcher_path = bindir / "gen_daqiri_config.py" + launcher_path.write_text(launcher, encoding="utf-8") + + source = tmp_path / "input.yaml" + source.write_text( + yaml.safe_dump(minimal_document(), sort_keys=False), + encoding="utf-8", + ) + + result = subprocess.run( + [sys.executable, str(launcher_path), "render", str(source)], + check=True, + capture_output=True, + text=True, + ) + assert yaml.safe_load(result.stdout) == minimal_document() + + +def test_raw_ibverbs_cli_requires_source_and_defaults_batch_size() -> None: + argv = [ + sys.executable, + str(REPOSITORY_ROOT / "scripts/gen_daqiri_config.py"), + "raw-pair", + "--tx-address", + "0000:01:00.0", + "--rx-address", + "0000:01:00.1", + "--master-core", + "1", + "--tx-queue-cores", + "2", + "--rx-queue-cores", + "3", + "--tx-worker-cores", + "4", + "--rx-worker-cores", + "5", + "--eth-dst-addr", + "02:00:00:00:00:02", + "--engine", + "ibverbs", + ] + with_source = subprocess.run( + [*argv, "--eth-src-addr", "02:00:00:00:00:01"], + check=True, + capture_output=True, + text=True, + ) + document = yaml.safe_load(with_source.stdout) + assert document["bench_tx"][0]["eth_src_addr"] == "02:00:00:00:00:01" + assert document["bench_tx"][0]["batch_size"] == 1024 + + without_source = subprocess.run(argv, capture_output=True, text=True) + assert without_source.returncode != 0 + assert ( + "eth_src_addr is required for ibverbs or engine-default benchmark profiles" + in without_source.stderr + ) + + +@pytest.mark.parametrize( + "transport,extra,error", + [ + ("roce", ["--rx-batch-size", "32"], "supported only for TCP/UDP"), + ("roce", ["--iterations", "10"], "supported only for TCP/UDP"), + ("udp", ["--rx-depth", "7"], "supported only for RoCE"), + ("udp", ["--tx-depth", "9"], "supported only for RoCE"), + ("tcp", ["--roce-transport-mode", "UD"], "supported only for RoCE"), + ], +) +def test_socket_cli_rejects_inapplicable_transport_options( + transport: str, extra: list[str], error: str +) -> None: + result = subprocess.run( + [ + sys.executable, + str(REPOSITORY_ROOT / "scripts/gen_daqiri_config.py"), + "socket-pair", + "--transport", + transport, + "--client-address", + "10.0.0.1", + "--server-address", + "10.0.0.2", + "--client-port", + "5002", + "--server-port", + "5001", + "--client-master-core", + "1", + "--server-master-core", + "2", + "--client-rx-core", + "3", + "--client-tx-core", + "4", + "--server-rx-core", + "5", + "--server-tx-core", + "6", + "--client-worker-core", + "7", + "--server-worker-core", + "8", + "--message-size", + "1024", + "--buffer-size", + "2048", + "--num-bufs", + "128", + "--role", + "tx", + *extra, + ], + capture_output=True, + text=True, + ) + assert result.returncode == 2 + assert error in result.stderr diff --git a/tests/portable/test_config_generator_profiles.py b/tests/portable/test_config_generator_profiles.py new file mode 100644 index 00000000..c471031b --- /dev/null +++ b/tests/portable/test_config_generator_profiles.py @@ -0,0 +1,292 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import pytest + +from scripts.daqiri_config import ( + ConfigError, + RawPairSpec, + SocketPairSpec, + generate_raw_pair, + generate_raw_roles, + generate_socket_pair, + render_document, +) + + +def socket_spec(transport: str) -> SocketPairSpec: + return SocketPairSpec( + transport=transport, + client_address="1.1.1.1", + server_address="2.2.2.2", + client_port=5002, + server_port=5001, + client_master_core=3, + server_master_core=4, + client_rx_core=5, + client_tx_core=6, + server_rx_core=7, + server_tx_core=8, + client_worker_core=9, + server_worker_core=10, + message_size=1024, + buffer_size=65536, + num_bufs=128, + rx_num_bufs=256 if transport == "roce" else None, + tx_num_bufs=64 if transport == "roce" else None, + rx_batch_size=32 if transport != "roce" else None, + tx_depth=64 if transport == "roce" else None, + ) + + +@pytest.mark.parametrize("transport", ["udp", "tcp", "roce"]) +def test_socket_pair_emits_independent_concrete_roles(transport: str) -> None: + documents = generate_socket_pair(socket_spec(transport)) + assert set(documents) == {"tx", "rx"} + for role, document in documents.items(): + assert "<" not in render_document(document) + config = document["daqiri"]["cfg"] + assert len(config["interfaces"]) == 1 + assert config["interfaces"][0]["socket_config"]["mode"] == ( + "client" if role == "tx" else "server" + ) + expected_prefix = "rdma_bench" if transport == "roce" else "socket_bench" + assert f"{expected_prefix}_client" in documents["tx"] + assert f"{expected_prefix}_server" in documents["rx"] + + +def test_roce_pair_preserves_direction_specific_flow_control_storage() -> None: + documents = generate_socket_pair(socket_spec("roce")) + for document in documents.values(): + regions = { + region["name"]: region["num_bufs"] + for region in document["daqiri"]["cfg"]["memory_regions"] + } + rx_name = next(name for name in regions if "_RX_" in name) + tx_name = next(name for name in regions if "_TX_" in name) + assert regions[rx_name] == 256 + assert regions[tx_name] == 64 + + +def raw_spec(**overrides) -> RawPairSpec: + values = { + "tx_address": "0000:01:00.0", + "rx_address": "0000:01:00.1", + "master_core": 3, + "tx_queue_cores": (4,), + "rx_queue_cores": (5,), + "tx_worker_cores": (6,), + "rx_worker_cores": (7,), + "eth_dst_addr": "02:00:00:00:00:02", + "eth_src_addr": "02:00:00:00:00:01", + "engine": "ibverbs", + } + values.update(overrides) + return RawPairSpec(**values) + + +@pytest.mark.parametrize("engine", ["dpdk", "ibverbs"]) +@pytest.mark.parametrize("transform", ["none", "vlan", "vxlan", "gre", "nvgre"]) +def test_raw_transform_matrix(transform: str, engine: str) -> None: + document = generate_raw_pair(raw_spec(transform=transform, engine=engine)) + tx = document["daqiri"]["cfg"]["interfaces"][0]["tx"] + if transform == "none": + assert "flows" not in tx + else: + assert tx["flows"][0]["actions"][0]["type"] in ( + "vlan_push", + "tunnel_encap", + ) + rx = document["daqiri"]["cfg"]["interfaces"][1]["rx"] + match = rx["flows"][0]["match"] + if engine == "dpdk": + assert match == {} + elif transform == "vlan": + assert match == {"udp_src": 4096, "udp_dst": 4096} + elif transform == "vxlan": + assert match == {"udp_src": 49152, "udp_dst": 4789} + else: + assert match == { + "ipv4_src": "192.0.2.1", + "ipv4_dst": "192.0.2.2", + } + + +@pytest.mark.parametrize("tx_count,rx_count", [(1, 1), (1, 2), (2, 1), (2, 2)]) +def test_raw_multi_queue_matrix_routes_every_flow(tx_count: int, rx_count: int) -> None: + document = generate_raw_pair( + raw_spec( + engine="dpdk", + memory_kind="host_pinned", + tx_queue_cores=tuple(range(10, 10 + tx_count)), + rx_queue_cores=tuple(range(20, 20 + rx_count)), + tx_worker_cores=tuple(range(30, 30 + tx_count)), + rx_worker_cores=tuple(range(40, 40 + rx_count)), + ) + ) + config = document["daqiri"]["cfg"] + assert len(config["memory_regions"]) == tx_count + rx_count + flows = config["interfaces"][1]["rx"]["flows"] + assert len(flows) == max(tx_count, rx_count) + assert [flow["action"]["id"] for flow in flows] == [ + index % rx_count for index in range(max(tx_count, rx_count)) + ] + if tx_count == 1 and rx_count == 2: + assert document["bench_tx"][0]["udp_dst_port"] == "4096-4097" + + +def test_raw_pair_can_split_for_cross_host_deployment() -> None: + documents = generate_raw_roles(raw_spec()) + assert list(documents["tx"]) == ["daqiri", "bench_tx"] + assert list(documents["rx"]) == ["daqiri", "bench_rx"] + assert len(documents["tx"]["daqiri"]["cfg"]["interfaces"]) == 1 + assert len(documents["rx"]["daqiri"]["cfg"]["interfaces"]) == 1 + + +def test_raw_buffer_size_and_tx_source_offload_are_configurable() -> None: + document = generate_raw_pair( + raw_spec(payload_size=64, buffer_size=8064, tx_eth_src=False) + ) + config = document["daqiri"]["cfg"] + assert {region["buf_size"] for region in config["memory_regions"]} == {8064} + tx_queue = config["interfaces"][0]["tx"]["queues"][0] + assert "offloads" not in tx_queue + + +def test_raw_batch_size_defaults_follow_backend_and_preserve_explicit_values() -> None: + dpdk = generate_raw_pair(raw_spec(engine="dpdk", eth_src_addr=None)) + ibverbs = generate_raw_pair(raw_spec(engine="ibverbs")) + engine_default = generate_raw_pair(raw_spec(engine=None)) + explicit = generate_raw_pair(raw_spec(engine="ibverbs", batch_size=2048)) + + assert ( + dpdk["daqiri"]["cfg"]["interfaces"][0]["tx"]["queues"][0]["batch_size"] + == 10240 + ) + assert dpdk["daqiri"]["cfg"]["interfaces"][1]["rx"]["queues"][0]["batch_size"] == 10240 + assert dpdk["bench_tx"][0]["batch_size"] == 10240 + assert ( + ibverbs["daqiri"]["cfg"]["interfaces"][0]["tx"]["queues"][0]["batch_size"] + == 1024 + ) + assert ibverbs["daqiri"]["cfg"]["interfaces"][1]["rx"]["queues"][0]["batch_size"] == 1024 + assert ibverbs["bench_tx"][0]["batch_size"] == 1024 + assert engine_default["bench_tx"][0]["batch_size"] == 1024 + assert explicit["bench_tx"][0]["batch_size"] == 2048 + + +@pytest.mark.parametrize("engine", ["ibverbs", None]) +def test_raw_ibverbs_benchmark_requires_source_address(engine: str | None) -> None: + with pytest.raises( + ConfigError, + match="eth_src_addr is required for ibverbs or engine-default benchmark profiles", + ): + raw_spec(engine=engine, eth_src_addr=None) + + +def test_raw_ibverbs_benchmark_emits_source_address_for_each_tx_queue() -> None: + document = generate_raw_pair( + raw_spec( + eth_src_addr="02:00:00:00:00:03", + tx_queue_cores=(4, 5), + tx_worker_cores=(6, 7), + ) + ) + assert [entry["eth_src_addr"] for entry in document["bench_tx"]] == [ + "02:00:00:00:00:03", + "02:00:00:00:00:03", + ] + + with pytest.raises(ConfigError, match="eth_src_addr must be a six-octet MAC address"): + raw_spec(eth_src_addr="invalid") + + +def test_raw_dpdk_source_address_is_optional_and_daqiri_only_needs_no_source() -> None: + dpdk = generate_raw_pair(raw_spec(engine="dpdk", eth_src_addr=None)) + production = generate_raw_pair( + raw_spec(engine="ibverbs", include_benchmark=False, eth_src_addr=None) + ) + assert "eth_src_addr" not in dpdk["bench_tx"][0] + assert "bench_tx" not in production + + +def test_invalid_profile_inputs_fail_before_rendering() -> None: + with pytest.raises(ConfigError, match="UDP message_size"): + socket_spec("udp").__class__( + **{**socket_spec("udp").__dict__, "message_size": 65508, "buffer_size": 65508} + ) + with pytest.raises(ConfigError, match="one tx_worker_core"): + raw_spec(tx_queue_cores=(4, 5), tx_worker_cores=(6,)) + with pytest.raises(ConfigError, match="explicit dpdk or ibverbs"): + raw_spec(transform="vlan", engine=None) + with pytest.raises(ConfigError, match="header_size must be at least 42"): + raw_spec(header_size=41) + with pytest.raises(ConfigError, match="IPv4 total-length"): + raw_spec(header_size=42, payload_size=65508) + with pytest.raises(ConfigError, match="runtime frame-size limit"): + raw_spec(header_size=42, payload_size=9059) + with pytest.raises(ConfigError, match="runtime frame-size limit"): + raw_spec(transform="vxlan", header_size=42, payload_size=9009) + with pytest.raises(ConfigError, match=r"at least header_size \+ payload_size"): + raw_spec(buffer_size=8000) + with pytest.raises(ConfigError, match="at least batch_size"): + raw_spec(engine="ibverbs", batch_size=2, num_bufs=1) + with pytest.raises(ConfigError, match="at least twice batch_size"): + raw_spec(engine="dpdk", batch_size=2, num_bufs=3) + with pytest.raises(ConfigError, match="at least twice batch_size"): + raw_spec(engine=None, batch_size=None, num_bufs=2047) + with pytest.raises(ConfigError, match="rx_batch_size must not exceed num_bufs"): + spec = socket_spec("udp") + spec.__class__(**{**spec.__dict__, "num_bufs": 16, "rx_batch_size": 32}) + with pytest.raises(ConfigError, match="supported only for RoCE"): + spec = socket_spec("udp") + spec.__class__(**{**spec.__dict__, "rx_num_bufs": 512}) + with pytest.raises( + ConfigError, match="rx_batch_size is supported only for TCP/UDP" + ): + spec = socket_spec("roce") + spec.__class__(**{**spec.__dict__, "rx_batch_size": 32}) + with pytest.raises( + ConfigError, match="iterations is supported only for TCP/UDP" + ): + spec = socket_spec("roce") + spec.__class__(**{**spec.__dict__, "iterations": 10}) + with pytest.raises(ConfigError, match="supported only for RoCE"): + spec = socket_spec("udp") + spec.__class__(**{**spec.__dict__, "rx_depth": 7}) + with pytest.raises(ConfigError, match="supported only for RoCE"): + spec = socket_spec("tcp") + spec.__class__(**{**spec.__dict__, "roce_transport_mode": "UD"}) + + +def test_production_profiles_do_not_require_benchmark_worker_cores() -> None: + raw = generate_raw_pair( + raw_spec( + include_benchmark=False, + tx_worker_cores=(), + rx_worker_cores=(), + header_size=1, + payload_size=1, + buffer_size=2048, + tx_eth_src=False, + ) + ) + assert "bench_tx" not in raw and "bench_rx" not in raw + raw_config = raw["daqiri"]["cfg"] + assert {region["buf_size"] for region in raw_config["memory_regions"]} == {2048} + assert "offloads" not in raw_config["interfaces"][0]["tx"]["queues"][0] + + spec = socket_spec("udp") + socket = generate_socket_pair( + spec.__class__( + **{ + **spec.__dict__, + "include_benchmark": False, + "client_worker_core": None, + "server_worker_core": None, + } + ) + ) + assert all(not any(key.startswith("socket_bench") for key in doc) for doc in socket.values()) diff --git a/tests/portable/test_config_validation_scripts.py b/tests/portable/test_config_validation_scripts.py new file mode 100644 index 00000000..86c8fece --- /dev/null +++ b/tests/portable/test_config_validation_scripts.py @@ -0,0 +1,224 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +from __future__ import annotations + +import os +import subprocess +import sys +from pathlib import Path + +import pytest + + +REPOSITORY_ROOT = Path(__file__).resolve().parents[2] +GENERATED_CHECK = REPOSITORY_ROOT / "scripts/check_generated_configs.py" +CHECKED_IN_CHECK = REPOSITORY_ROOT / "scripts/check_daqiri_configs.py" + +FAKE_VALIDATOR = """\ +#!/usr/bin/env python3 +import os +import sys +from pathlib import Path + +log = Path(os.environ["FAKE_VALIDATOR_LOG"]) +if sys.argv[1:] == ["--list-engines"]: + with log.open("a", encoding="utf-8") as stream: + stream.write("QUERY\\n") + if os.environ.get("FAKE_VALIDATOR_QUERY_FAILURE"): + raise SystemExit(7) + print(os.environ["FAKE_VALIDATOR_ENGINES"]) + raise SystemExit(0) + +invalid_markers = ( + "batch_sise", + "4294967296", + "tx_eth_scr", + "flows: typo", + "ecpri_typo", + "ipv4_src:\\n", + "devcie", + "locla", + 'log_level: "verbose"', + "not-an-ip", + "5001junk", + "missing-region", + "id: 999", + "duplicate_flow_id", + "UNSUPPORTED_MARKER", +) +return_code = 0 +for argument in sys.argv[1:]: + path = Path(argument) + text = path.read_text(encoding="utf-8") + kind = "invalid" if any(marker in text for marker in invalid_markers) else "valid" + with log.open("a", encoding="utf-8") as stream: + stream.write(f"VALIDATE\\t{path.name}\\t{kind}\\n") + if kind == "invalid": + return_code = 1 +raise SystemExit(return_code) +""" + + +def fake_validator(tmp_path: Path) -> tuple[Path, Path]: + validator = tmp_path / "fake-validator" + validator.write_text(FAKE_VALIDATOR, encoding="utf-8") + validator.chmod(0o755) + log = tmp_path / "validator.log" + return validator, log + + +def run_script( + script: Path, + validator: Path, + log: Path, + engines: str, + *extra: str, + query_failure: bool = False, +) -> subprocess.CompletedProcess[str]: + environment = os.environ.copy() + environment["FAKE_VALIDATOR_ENGINES"] = engines + environment["FAKE_VALIDATOR_LOG"] = str(log) + if query_failure: + environment["FAKE_VALIDATOR_QUERY_FAILURE"] = "1" + return subprocess.run( + [sys.executable, str(script), "--validator", str(validator), *extra], + cwd=REPOSITORY_ROOT, + env=environment, + capture_output=True, + text=True, + check=False, + ) + + +def validation_names(log: Path) -> list[str]: + return [ + line.split("\t", 2)[1] + for line in log.read_text().splitlines() + if line.startswith("VALIDATE") + ] + + +@pytest.mark.parametrize( + ("engines", "required_present", "required_absent"), + [ + ( + "socket dpdk ibverbs", + { + "socket-udp-tx.yaml", + "socket-tcp-tx.yaml", + "socket-roce-tx.yaml", + "raw-dpdk-none.yaml", + "raw-ibverbs-none.yaml", + "raw-xhost-tx.yaml", + }, + set(), + ), + ( + "socket dpdk", + {"socket-udp-tx.yaml", "socket-tcp-tx.yaml", "raw-dpdk-none.yaml"}, + {"socket-roce-tx.yaml", "raw-ibverbs-none.yaml", "raw-xhost-tx.yaml"}, + ), + ( + "socket ibverbs", + { + "socket-udp-tx.yaml", + "socket-tcp-tx.yaml", + "socket-roce-tx.yaml", + "raw-ibverbs-none.yaml", + "raw-xhost-tx.yaml", + }, + {"raw-dpdk-none.yaml"}, + ), + ( + "socket", + {"socket-udp-tx.yaml", "socket-tcp-tx.yaml"}, + {"socket-roce-tx.yaml", "raw-dpdk-none.yaml", "raw-ibverbs-none.yaml"}, + ), + ], +) +def test_generated_check_selects_only_supported_engine_profiles( + tmp_path: Path, + engines: str, + required_present: set[str], + required_absent: set[str], +) -> None: + validator, log = fake_validator(tmp_path) + result = run_script(GENERATED_CHECK, validator, log, engines) + + assert result.returncode == 0, result.stderr + names = set(validation_names(log)) + assert required_present <= names + assert names.isdisjoint(required_absent) + + +def test_generated_check_keeps_exclude_dpdk_as_an_additional_restriction( + tmp_path: Path, +) -> None: + validator, log = fake_validator(tmp_path) + result = run_script( + GENERATED_CHECK, validator, log, "socket dpdk ibverbs", "--exclude-dpdk" + ) + + assert result.returncode == 0, result.stderr + names = set(validation_names(log)) + assert not any(name.startswith("raw-dpdk-") for name in names) + assert not any(name.startswith("raw-mq-") for name in names) + assert "raw-ibverbs-none.yaml" in names + + +@pytest.mark.parametrize("script", [GENERATED_CHECK, CHECKED_IN_CHECK]) +def test_check_fails_when_capability_query_fails(tmp_path: Path, script: Path) -> None: + validator, log = fake_validator(tmp_path) + result = run_script( + script, + validator, + log, + "socket dpdk ibverbs", + query_failure=True, + ) + + assert result.returncode == 1 + assert "Cannot determine the validator's compiled engines." in result.stderr + + +@pytest.mark.parametrize( + "engines", ["socket", "socket dpdk", "socket ibverbs", "socket dpdk ibverbs"] +) +def test_checked_in_default_selects_supported_cases(tmp_path: Path, engines: str) -> None: + validator, log = fake_validator(tmp_path) + result = run_script(CHECKED_IN_CHECK, validator, log, engines) + + assert result.returncode == 0, result.stderr + names = validation_names(log) + assert any(name.endswith("daqiri_bench_socket_udp_tx_rx.yaml") for name in names) + assert any( + name.endswith("daqiri_bench_raw_hw_loopback_ibverbs.yaml") for name in names + ) == ("ibverbs" in engines) + assert any( + name.endswith("daqiri_bench_raw_sw_loopback.yaml") for name in names + ) == ("dpdk" in engines) + assert any( + name.endswith("daqiri_bench_raw_tx_rx.yaml") for name in names + ) == ("dpdk" in engines or "ibverbs" in engines) + assert ("zero-flow-id-per-interface.yaml" in names) == ("dpdk" in engines) + assert "unknown-queue-key.yaml" in names + assert "malformed-tx-flows.yaml" in names + assert ("unknown-reorder-flow.yaml" in names) == ("ibverbs" in engines) + + +def test_checked_in_explicit_path_is_unfiltered_and_validator_failure_is_reported( + tmp_path: Path, +) -> None: + validator, log = fake_validator(tmp_path) + config = tmp_path / "unsupported.yaml" + config.write_text( + "daqiri:\n cfg:\n stream_type: raw\n engine: dpdk\n" + " UNSUPPORTED_MARKER: true\n", + encoding="utf-8", + ) + + result = run_script(CHECKED_IN_CHECK, validator, log, "socket", str(config)) + + assert result.returncode == 1 + assert any(line.endswith("\tinvalid") for line in log.read_text().splitlines()) diff --git a/tests/requirements.txt b/tests/requirements.txt index c99fd184..b275b7e2 100644 --- a/tests/requirements.txt +++ b/tests/requirements.txt @@ -2,3 +2,4 @@ # SPDX-License-Identifier: Apache-2.0 pytest>=8,<10 +PyYAML>=6,<7 diff --git a/tools/config_validate.cpp b/tools/config_validate.cpp index 49aaaa27..4f06c9b9 100644 --- a/tools/config_validate.cpp +++ b/tools/config_validate.cpp @@ -9,10 +9,23 @@ #include "src/engine.h" #include +#include int main(int argc, char** argv) { + if (argc == 2 && std::string(argv[1]) == "--list-engines") { + std::cout << "socket"; +#if DAQIRI_ENGINE_DPDK + std::cout << " dpdk"; +#endif +#if DAQIRI_ENGINE_IBVERBS || DAQIRI_ENGINE_RDMA + std::cout << " ibverbs"; +#endif + std::cout << '\n'; + return 0; + } + if (argc < 2) { - std::cerr << "Usage: daqiri_config_validate [...]\n"; + std::cerr << "Usage: daqiri_config_validate [...] | --list-engines\n"; return 2; }