From b39576f5ca68cc5f175e8d4db636a2c972ccc5a9 Mon Sep 17 00:00:00 2001 From: hbarclay Date: Mon, 28 Sep 2026 00:03:17 -0400 Subject: [PATCH 1/6] operatorx: multi-device gemm, moe and attention under InferenceX recipes Co-Authored-By: Claude Opus 5.5 (1M context) --- operatorx/CI.md | 193 +- operatorx/CI_zh.md | 110 +- operatorx/CLUSTERS.md | 111 - operatorx/README.md | 80 +- operatorx/ci.py | 145 +- operatorx/clusters.py | 6 +- operatorx/core/backend.py | 16 +- operatorx/core/errors.py | 7 +- operatorx/core/op.py | 9 +- operatorx/core/op_registry.py | 6 +- operatorx/core/parallel.py | 44 + operatorx/core/result.py | 11 +- operatorx/main.py | 79 +- operatorx/ops/attention.py | 41 +- operatorx/ops/gemm.py | 25 +- operatorx/ops/moe.py | 27 +- operatorx/recipes.py | 34 +- operatorx/runners/amd/backends/torch.py | 2 +- operatorx/runners/amd/backends/vllm.py | 5 +- operatorx/runners/amd/runner.py | 54 +- operatorx/runners/common/collectivex.py | 23 + operatorx/runners/common/profiling.py | 51 +- operatorx/runners/common/ranks.py | 100 + operatorx/runners/common/telemetry.py | 50 +- operatorx/runners/common/timing.py | 56 + operatorx/runners/common/trace.py | 18 +- operatorx/runners/common/vllm/__init__.py | 2 +- operatorx/runners/common/vllm/attention.py | 158 +- operatorx/runners/common/vllm/engine.py | 135 + operatorx/runners/common/vllm/linear.py | 195 +- operatorx/runners/common/vllm/modules.py | 72 + operatorx/runners/common/vllm/moe.py | 235 +- operatorx/runners/nvidia/backends/vllm.py | 2 +- operatorx/runners/nvidia/runner.py | 59 +- operatorx/runners/nvidia/runtime.py | 4 +- operatorx/runtime.py | 20 +- operatorx/scripts/consolidate_results.py | 19 +- operatorx/scripts/pull_containers.py | 98 - operatorx/scripts/submit_run.py | 192 - operatorx/testlists/attn_parallel.json | 11884 +++++++++++++++++++ operatorx/testlists/gemm_parallel.json | 1692 +++ operatorx/testlists/moe_parallel.json | 5680 +++++++++ operatorx/tests/test_ci.py | 15 +- 43 files changed, 20284 insertions(+), 1481 deletions(-) delete mode 100644 operatorx/CLUSTERS.md create mode 100644 operatorx/core/parallel.py create mode 100644 operatorx/runners/common/collectivex.py create mode 100644 operatorx/runners/common/ranks.py create mode 100644 operatorx/runners/common/timing.py create mode 100644 operatorx/runners/common/vllm/engine.py create mode 100644 operatorx/runners/common/vllm/modules.py delete mode 100755 operatorx/scripts/pull_containers.py delete mode 100755 operatorx/scripts/submit_run.py create mode 100644 operatorx/testlists/attn_parallel.json create mode 100644 operatorx/testlists/gemm_parallel.json create mode 100644 operatorx/testlists/moe_parallel.json diff --git a/operatorx/CI.md b/operatorx/CI.md index 2ff613f804..bf711ca3f6 100644 --- a/operatorx/CI.md +++ b/operatorx/CI.md @@ -2,16 +2,11 @@ **English** | [中文](CI_zh.md) -[OperatorX Sweep](../../.github/workflows/operatorx-sweep.yml) runs manually on the -self-hosted runners carrying one of these labels, the labels InferenceX's benchmark -workflows use: `cluster:h100-dgxc` (default), `cluster:h200-dgxc`, `cluster:b200-nscale`, -`cluster:b300-dsxe`, `cluster:gb200-nv`, `cluster:gb300-nv`, `cluster:mi300x-amd`, -`cluster:mi325x-amds` or `cluster:mi355x-amds`. -Pull requests only run the hosted planner; GPU work requires `workflow_dispatch`. +[OperatorX Sweep](../.github/workflows/operatorx-sweep.yml) runs on `workflow_dispatch` only (pull requests run just the hosted planner). -| GPU | Runner label | GPUs per physical node | Image platform | Result cluster | +| GPU | Runner label | GPUs / node | Image platform | Result cluster | | --- | --- | ---: | --- | --- | -| H100 | `cluster:h100-dgxc` | 8 | `linux/amd64` | `h100_dgxc_8x` | +| H100 | `cluster:h100-dgxc` (default) | 8 | `linux/amd64` | `h100_dgxc_8x` | | H200 | `cluster:h200-dgxc` | 8 | `linux/amd64` | `h200_dgxc_8x` | | B200 | `cluster:b200-nscale` | 8 | `linux/amd64` | `b200_nscale_8x` | | B300 | `cluster:b300-dsxe` | 8 | `linux/amd64` | `b300_dsxe_8x` | @@ -21,190 +16,30 @@ Pull requests only run the hosted planner; GPU work requires `workflow_dispatch` | MI325X | `cluster:mi325x-amds` | 8 | `linux/amd64` | `mi325x_amds_8x` | | MI355X | `cluster:mi355x-amds` | 8 | `linux/amd64` | `mi355x_8x` | -GB200/GB300 runs use one four-GPU tray, not the full NVL72 rack. Dense GEMM uses -`world_sizes=1` on every runner; reported TFLOPS remains per GPU. Hardware facts come -from CollectiveX's platform registry, and both planning and execution validate them. - ## Dispatch -Once GitHub has registered the workflow, select **OperatorX Sweep → Run workflow**, -choose the source branch, and keep the initial defaults: `runner=cluster:h100-dgxc`, -`backends=vllm`, `testlists=gemm`, `world_sizes=1`, `chunk_size=500`. -This schedules the complete checked-in GEMM catalog in bounded shards (currently -5,416 cases in 11 shards). The catalog includes formats unsupported by a selected -backend and shapes that can exceed device memory. Unsupported rows remain visible; -actual kernel and allocation errors fail CI. A full catalog run is not a promise -that every case fits or is supported on H100. A newly added workflow -may need to reach the default branch before GitHub accepts manual dispatch. - ```bash gh workflow run operatorx-sweep.yml --repo SemiAnalysisAI/InferenceX \ --ref -f runner=cluster:h100-dgxc -f backends=vllm \ - -f testlists=gemm -f world_sizes=1 -f chunk_size=500 + -f testlists=gemm -f world_sizes=1 -f chunk_size=500 -f ingest=false ``` -`mode=timing` (default) records latency, telemetry and a profiler replay per op. -`mode=counters` instead runs each op once under Nsight Compute (NVIDIA; the host's -`/opt/nvidia/nsight-compute`) or rocprofv3 (AMD; 12 counter passes, so use a smaller -`chunk_size`), with raw counter files under `results/counters/`. Latencies from a -counters run are perturbed by the profiler. - -For a quick infrastructure smoke check, explicitly select `testlists=gemm_perf` -and `chunk_size=50` (11 BF16 cases). `attn_deepseek_v4`, `attn_kimi_k3` and `attn_minimax_m3` hold attention modules; dispatch -them separately from gemm / moe lists. Test runs set `-f ingest=false` so their results stay -out of the OperatorX database; a dispatched run is ingested by default. `gemm_serving_8k1k_min` and -`gemm_serving_all_min` hold the GEMMs of InferenceX serving configurations. Unsupported operations remain visible in results. Backend -import errors, benchmark errors, and zero successful rows fail the shard. -Start with BF16 GEMM, then the quantized formats. -Do not infer that Blackwell-specific FP4 kernels work on Hopper. +- Inputs: `runner`, `backends` (`vllm`; AMD also `torch`), `testlists`, `mode` (`timing` | `counters`), `world_sizes` (1, 2, 4, 8), `chunk_size` (1-500), `recovery_run_id`, `ingest` (default true; `false` for test runs). +- `mode=counters`: each op once under Nsight Compute (NVIDIA) or rocprofv3 (AMD, 12 passes: use a smaller `chunk_size`); raw files in `results/counters/`; latencies perturbed. +- Smoke check: `testlists=gemm_perf`, `chunk_size=50`. Dispatch `attn_*` testlists apart from gemm / moe ones. ## Execution contract -- Hosted planning validates inputs, groups backends by container image, separates - world sizes and MoE parallelism triples, and splits shapes into bounded chunks. - At most 256 shards are accepted. World sizes are restricted to 1, 2, 4, and 8, - and must fit one physical node (GB200/GB300 reject 8). Shapes outside the - requested sizes are counted in `excluded_shapes`. -- Each Actions shard holds exactly one exclusive physical Slurm node (four or eight GPUs). The GPU - process count is the selected world size. Admission uses the existing priority - scorer and `ci-job-*`, `ci-attempt-*`, and exactly one `nodes:1` label. All shards remain eligible; the priority/node scheduler controls physical-node - admission. A second GitHub matrix cap can strand assigned labels on held jobs. - Both scheduler switches must remain enabled. -- Runner settings come from CollectiveX's tracked platform registry. Source is - checked out at the workflow SHA and copied into a private, compute-visible - directory below the configured shared squash parent or a writable configured - `storage_roots` entry (GB200). B300 uses the compute-visible account home from - the password database, matching CollectiveX; an explicit `stage_dir` takes - precedence. Results never depend on - a submit-host `/tmp` mount being visible to compute nodes. -- The planner resolves each image digest. Imports are locked and cached by image - plus digest and CPU architecture, with a second digest check after import. A moved or unresolvable - tag fails rather than claiming the planned image was measured. Images must be - anonymously readable from the planning and import hosts. Imports verify the host - CPU architecture and run inside their allocation. B300 follows the inference - launcher's compute-node import because its submit host lacks extraction space - for this image. Enroot and GNU parallel use private temporary directories; - Enroot uses explicit registry URLs and any platform-configured cache path. Allocation - forwards account, QoS, and quarantined nodes; B300/GB nodes retain their existing - remap-root and memory settings. B300 leaves QoS selection to its partition/account, - matching the inference launcher; the former `batch_1_qos` override is rejected - by the current cluster. Its former excluded node names also do not exist in - this cluster and have been removed; Slurm still honors drained nodes. GB300 retains - its configured QoS and exclusions. -- The launcher remains active through allocation, import, and execution. The - allocation time limit is 45 minutes; Actions permits 70 minutes including - queueing and cleanup. Slurm job names match the Actions runner name. -- Signals and the workflow's `always()` recovery step cancel recorded allocations, - stop writers, recover partial results, and remove staged sources. The workflow - explicitly allows 180 seconds for Slurm epilog/node release, including on H200. A failed - cleanup retains staging for investigation. Slurm's time limit is the last - bound if the runner host disappears. -- Strict CI runs atomically checkpoint rank-zero rows after each operation, - outside kernel timing. The existing non-CI timing loop is unchanged. - -## Artifacts and reruns - -`operatorx-manifest-` records requested cases, image digests and source -SHA. It remains available to failed-job reruns. Each attempt uploads separate -`operatorx-shard---` artifacts containing execution -metadata, allocation/import/benchmark logs, status and any raw result JSONs. -Startup failures can have logs without results; cancellation checkpoints are -partial coverage. A successful shard requires successful measurements, not just -successful Slurm submission. Result environments record the workflow run, -attempt, shard, source SHA and image digest. - -Download artifacts through `gh run download`. Preserve raw files and provenance; -`scripts/consolidate_results.py` is not part of CI. +- Planning: shards per backend image, `parallel` split and InferenceX recipe (`recipes.py`: image, launch env, `vllm serve` args), chunked by `chunk_size`; at most 256 shards; a case with no recipe runs on the backend's image from `containers.toml`. +- One exclusive Slurm node per shard (GB200/GB300: one 4-GPU tray, no world size 8); 45 min allocation, 70 min job. +- Image: digest resolved at planning, imported with enroot on the allocated node, cached by image + digest + arch; a tag that moved since planning fails the shard. +- Ranks: one process per GPU of the split's world size, env:// rendezvous; runner settings from CollectiveX's platform registry overlaid by `platforms.json`. +- Results: rows checkpointed after each op; unsupported cases stay visible as rows, errors or zero successful rows fail the shard; artifacts `operatorx-manifest-` and `operatorx-shard---`; ingested into the OperatorX database unless `ingest=false`. +- Cleanup: cancellation and the `always()` step cancel the allocation and keep partial results; after a failed cleanup, rerun one shard with `recovery_run_id`. -## Local validation - -Planning requires Python 3.11 or newer. Compute-side control code uses the existing -Python 3.10+ Slurm-host environment. CPU tests run the real planner, benchmark -orchestration and launcher with external GPU/Slurm collaborators substituted. +## Local tests ```bash uv run --no-project --python 3.12 --with pytest --with pyyaml --with torch --with numpy \ python -m pytest operatorx/tests/ -q ``` - -Real acceptance additionally requires a smoke run with artifacts on each selected runner, a -failed-shard rerun, and cancellation with confirmed allocation release. CPU -checks alone do not establish GPU compatibility or cluster storage visibility. - -The final coverage job selects the newest artifact attempt for each requested -shard, preserves successful shards from previous attempts, and fails if any shard -is missing or failed. Its summary separates requested shapes from result rows -(one shape may run on multiple backends). - -If cleanup failed, a single-shard dispatch can set `recovery_run_id` to the recent -OperatorX run from the same runner. It downloads the execution artifacts and retries -allocation/staging cleanup before allocating a new node. Recovery checks the run, -runner, and private staging parent; do not select unrelated or old Slurm executions. - -`cleanup.log` records the active-job query used to confirm allocation release. -It queries the current user’s job list because querying a removed job ID directly -can return a Slurm error even after that allocation has terminated. - -## AMD execution - -`platforms.json` overlays the CollectiveX registry, per runner label, with the AMDS Slurm clusters. -AMD accepts single-GPU `torch`/`vllm` GEMM. ROCm PyTorch uses HIP events through -`torch.cuda`; FP8 selects FNUZ on gfx942 and OCP on gfx950. Unsupported formats -remain explicit. Staging lives outside `_work`, below the shared runner root -derived from `RUNNER_TEMP`. Containers never write to the checkout. MI300X/MI325X -forward `/dev/kfd` and `/dev/dri`; CPU requests follow each inference launcher. - -## GEMM - -`gemm` args describe each operand's storage and quantization. `a` is the -activation `[M, K]` and `b` the weight `[N, K]`: `{"dtype", "scale"?, "scale2"?, -"symmetric"?}`, where `scale` is `{"dtype", "static", "group": [rows, cols]}` and -`-1` spans a dimension (`[-1, -1]` per-tensor, `[1, -1]` per-token, `[-1, 1]` -per-channel, `[1, 128]` 1x128 groups, `[128, 128]` blocks). For example, FP8 -block quantization is -`"a": {"dtype": "e4m3", "scale": {"dtype": "fp32", "static": false, "group": [1, 128]}}` and -`"b": {"dtype": "e4m3", "scale": {"dtype": "fp32", "static": true, "group": [128, 128]}}`; -an unquantized operand is `{"dtype": "bf16"}`. `scale.dtype` is the checkpoint's -scale format. `input` is the dtype an operand arrives in (default bf16): when it -differs from `dtype`, quantization is part of the op; when equal, the operand is -pre-quantized. The runtime format a framework converts to is reported per result. - -The `vllm` backend builds a vLLM `ReplicatedLinear` under the checkpoint quant -config those descriptors imply, runs `process_weights_after_loading` and times -`layer(x)`, so vLLM picks the kernel. `metrics.backend_meta` records the chosen -kernel classes, parameter dtypes before and after loading, and the vLLM env. -Emulation-only paths are reported unsupported. AMD enables AITER, as -InferenceX's ROCm launches do. - -## MoE - -`moe` (`ops/moe.py`) describes one MoE layer from the router GEMM on -normed hidden states to the combined output: shape (`tokens`, `hidden`), -routed `experts` (count, top-k, intermediate size, optional biases / latent width / -zero experts, and gemm operand descriptors `a1`, `w1`, `w2`, `a2`), `router` -(gate dtype, scoring, top-k / grouped / hash selection, bias, renormalize, scale), -`activation`, optional `shared` experts, and the `routing` data distribution. -Execution (expert kernels, dispatch, shared-expert fusion or stream overlap, graphs) -is the backend's choice. The `vllm` backend (`runners/common/vllm/moe.py`) builds -vLLM's router (`GateLinear`), routed experts (`FusedMoEFactory`) and shared-expert -MLP under the quant configs the descriptors imply and times the block; vLLM picks the -expert kernels, shared-expert fusion and streams. A forced expert-load distribution -(`balanced`, `zipf`, `single_hot`) replaces the router's expert choice and keeps its -weights; zero experts are reported unsupported for now. - -`testlists/moe_small.json` holds one full-size routed MoE layer per InferenceX -MoE checkpoint scheme (DeepSeek-R1, DeepSeek-V4-Pro/V4.1-Flash, Qwen3.5, Qwen3.8-Flash-Next, -GLM-5.2, Kimi-K3, MiniMax-M3 in their FP8/NVFP4/MXFP4/MXFP8 variants) at `tokens=1`, -plus `tokens=256` layers under each expert-load distribution, 28 cases. Each layer's -weights fit on one GPU. - -Every testlist entry says where it comes from: `sources` lists one -`/` per layer that runs the case, e.g. -`deepseek-ai/DeepSeek-V4-Pro/attn.wq_b` or `zai-org/GLM-5-FP8/mlp`. A checkpoint id is -`org/model`, so the role is what follows the last `/`. A shape shared by several models, -or by several layers of one model, lists every pair, so each role stays tied to its -model. The list is empty for a shape from no model. The loader rejects an entry without -it, and it is carried into each result's `op`; it is not part of the op's identity. - -Experimental operator changes are recorded in the adjacent `perf-changelog.yaml`, -separately from the root inference-recipe changelog's config-key schema. diff --git a/operatorx/CI_zh.md b/operatorx/CI_zh.md index cc05e0ffc3..5bd5fe6d98 100644 --- a/operatorx/CI_zh.md +++ b/operatorx/CI_zh.md @@ -2,15 +2,11 @@ [English](CI.md) | **中文** -[OperatorX Sweep](../../.github/workflows/operatorx-sweep.yml) 在带有以下标签之一的自托管运行器上手动运行(与 InferenceX 基准工作流使用的标签一致): -`cluster:h100-dgxc`(默认)、`cluster:h200-dgxc`、`cluster:b200-nscale`、 -`cluster:b300-dsxe`、`cluster:gb200-nv`、`cluster:gb300-nv`、`cluster:mi300x-amd`、 -`cluster:mi325x-amds` 或 `cluster:mi355x-amds`。 -PR 只在 GitHub 托管运行器上生成执行计划;GPU 执行必须通过 `workflow_dispatch` 触发。 +[OperatorX Sweep](../.github/workflows/operatorx-sweep.yml) 只通过 `workflow_dispatch` 运行(PR 只运行托管规划器)。 -| GPU | 运行器标签 | 每个物理节点的 GPU 数 | 镜像平台 | 结果集群标识 | +| GPU | 运行器标签 | 每节点 GPU 数 | 镜像平台 | 结果集群标识 | | --- | --- | ---: | --- | --- | -| H100 | `cluster:h100-dgxc` | 8 | `linux/amd64` | `h100_dgxc_8x` | +| H100 | `cluster:h100-dgxc`(默认) | 8 | `linux/amd64` | `h100_dgxc_8x` | | H200 | `cluster:h200-dgxc` | 8 | `linux/amd64` | `h200_dgxc_8x` | | B200 | `cluster:b200-nscale` | 8 | `linux/amd64` | `b200_nscale_8x` | | B300 | `cluster:b300-dsxe` | 8 | `linux/amd64` | `b300_dsxe_8x` | @@ -20,108 +16,30 @@ PR 只在 GitHub 托管运行器上生成执行计划;GPU 执行必须通过 ` | MI325X | `cluster:mi325x-amds` | 8 | `linux/amd64` | `mi325x_amds_8x` | | MI355X | `cluster:mi355x-amds` | 8 | `linux/amd64` | `mi355x_8x` | -GB200/GB300 每次使用一个四卡计算托盘,不会占用整个 NVL72 机架。所有运行器的 -稠密 GEMM 都使用 `world_sizes=1`,TFLOPS 始终按单卡计算。硬件信息来自 CollectiveX -平台配置,并在规划和执行阶段分别校验。 - ## 触发运行 -GitHub 注册该工作流后,选择 **OperatorX Sweep → Run workflow**,指定源码分支, -并保留初始默认值:`runner=cluster:h100-dgxc`、`backends=vllm`、`testlists=gemm`、 -`world_sizes=1`、`chunk_size=500`。这会将完整的 GEMM 测试列表拆分为有界分片(目前为 5,416 个测试、11 个分片)。 -列表中包含所选后端不支持的精度,以及可能超出设备显存的形状。不支持的测试会保留 -在结果中;实际的内核和显存分配错误仍会使 CI 失败。运行完整列表不代表其中每个 -测试都能在 H100 上执行。 -新工作流可能需要先进入默认分支,GitHub 才允许手动触发。 - ```bash gh workflow run operatorx-sweep.yml --repo SemiAnalysisAI/InferenceX \ --ref -f runner=cluster:h100-dgxc -f backends=vllm \ - -f testlists=gemm -f world_sizes=1 -f chunk_size=500 + -f testlists=gemm -f world_sizes=1 -f chunk_size=500 -f ingest=false ``` -快速检查基础设施时,可显式选择 `testlists=gemm_perf` 和 `chunk_size=50` -(11 个 BF16 测试)。`attn_deepseek_v4`、`attn_kimi_k3` 和 `attn_minimax_m3` 包含注意力模块,需与 gemm / moe 测试列表分开触发。测试运行应设置 `-f ingest=false`,使结果不进入 OperatorX 数据库;手动触发的运行默认会被导入。`gemm_serving_8k1k_min` 和 `gemm_serving_all_min` -包含 InferenceX 推理服务配置中的 GEMM。 -不支持的操作会保留在结果中。后端导入错误、基准错误,以及没有任何成功结果, -都会使分片失败。先验证 BF16 GEMM,再验证量化格式。 -不要假定面向 Blackwell 的 FP4 内核可以在 Hopper 上运行。 +- 输入:`runner`、`backends`(`vllm`;AMD 另有 `torch`)、`testlists`、`mode`(`timing` | `counters`)、`world_sizes`(1、2、4、8)、`chunk_size`(1-500)、`recovery_run_id`、`ingest`(默认 true;测试运行设为 `false`)。 +- `mode=counters`:每个算子在 Nsight Compute(NVIDIA)或 rocprofv3(AMD,12 轮采集:请用更小的 `chunk_size`)下运行一次;原始文件在 `results/counters/`;延迟受分析器干扰。 +- 冒烟测试:`testlists=gemm_perf`、`chunk_size=50`。`attn_*` 测试列表与 gemm / moe 列表分开触发。 ## 执行约定 -- 托管规划步骤校验输入,按容器镜像分组后端,区分 world size 和 MoE 并行参数组合, - 并将形状拆成大小受限的分片。最多支持 256 个分片。world size 仅允许 1、2、4、8, - 且不能超过一个物理节点的 GPU 数,因此 GB200/GB300 不接受 8。 - 未被所选大小覆盖的形状会计入 `excluded_shapes`。 -- 每个 Actions 分片独占一个物理 Slurm 节点,包含四张或八张 GPU。GPU 进程数等于所选 world size。 - 准入沿用优先级评分器,以及 `ci-job-*`、`ci-attempt-*` 和唯一的 `nodes:1` 标签。 - 所有分片均可被调度,由优先级/节点调度器控制物理节点准入。额外的 GitHub matrix - 并发上限可能将标签分配给尚未获准启动的作业,导致停滞。两个调度开关都必须保持启用。 -- 运行器设置来自 CollectiveX 已纳入版本控制的平台配置。源码按工作流 SHA 检出, - 再复制到共享 squash 父目录或配置中可写的 `storage_roots` 路径(GB200)下的私有目录, - 该目录必须在计算节点上可见。B300 沿用 CollectiveX,从系统账户数据库读取计算节点可见 - 的账户主目录;显式配置的 `stage_dir` 优先。结果不依赖提交主机的 `/tmp` 在计算节点上可见。 -- 规划步骤解析镜像 digest。导入操作加锁,并按镜像、digest 和 CPU 架构缓存,导入后再次核对 - digest。标签发生变化或无法解析时运行失败,避免错误标注测量所用镜像。 - 规划和导入主机都必须能匿名读取镜像。导入前校验主机 CPU 架构,并在已分配的计算节点 - 上执行。B300 提交主机缺少该镜像所需的解压空间,因此沿用推理启动器的计算节点导入方式。 - Enroot 和 GNU parallel 使用私有临时目录;Enroot 使用显式 registry 地址及平台配置指定的 - 缓存路径。分配请求保留 account、QoS 和隔离节点 - 列表,并沿用 B300/GB 平台的 remap-root 与内存设置。B300 与推理启动器一致,由 - partition/account 选择 QoS;原来的 `batch_1_qos` 覆盖值会被当前集群拒绝。 - 原配置中的隔离节点名称也不存在于该集群,已移除;Slurm 仍会遵循节点的 drain - 状态。GB300 继续使用其配置的 QoS 和隔离节点列表。 -- 启动器等待分配、导入和执行完成。Slurm 分配限时 45 分钟;Actions 允许 70 分钟, - 包含排队与清理时间。Slurm 作业名与 Actions 运行器名称一致。 -- 信号处理和工作流的 `always()` 恢复步骤会取消已记录的分配、停止写入、保留部分结果, - 然后删除暂存源码。工作流显式允许最多 180 秒等待 Slurm epilog 和节点释放,覆盖 H200 - 的延迟释放情况。清理失败时保留暂存目录供调查。如果运行器主机失联,Slurm 时间限制 - 是最后的资源释放保障。 -- CI 严格模式在每个操作结束后原子写入 rank-zero 结果检查点,写入发生在内核计时之外。 - 原有非 CI 计时循环保持不变。 - -## 产物与重跑 - -`operatorx-manifest-` 记录请求的案例、镜像 digest 和源码 SHA,失败作业重跑时 -仍可使用。每次尝试分别上传 `operatorx-shard---`,包含执行元数据、 -分配/导入/基准日志、状态和已生成的原始结果 JSON。 -启动失败时可能只有日志;取消时的检查点仅代表部分覆盖。分片成功要求实际测量成功, -不能仅凭 Slurm 提交成功。结果环境信息记录工作流运行、尝试、分片、源码 SHA 和镜像 digest。 +- 规划:按后端镜像、`parallel` 拆分和 InferenceX 配方(`recipes.py`:镜像、启动环境变量、`vllm serve` 参数)分片,并按 `chunk_size` 切块;最多 256 个分片;没有配方的测试项使用 `containers.toml` 中的后端镜像。 +- 每个分片独占一个 Slurm 节点(GB200/GB300:一个四卡托盘,不支持 world size 8);分配时限 45 分钟,作业时限 70 分钟。 +- 镜像:规划时解析摘要,在分配到的节点上用 enroot 导入,按镜像 + 摘要 + 架构缓存;规划后标签发生变化则分片失败。 +- Rank:拆分的 world size 中每块 GPU 一个进程,通过 env:// 汇合;运行器设置来自 CollectiveX 平台配置,并由 `platforms.json` 覆盖。 +- 结果:每个算子后保存检查点;不支持的测试项作为结果行保留,错误或没有成功行会使分片失败;产物为 `operatorx-manifest-` 和 `operatorx-shard---`;除非 `ingest=false`,结果会导入 OperatorX 数据库。 +- 清理:取消和 `always()` 步骤会取消分配并保留部分结果;清理失败后,用 `recovery_run_id` 重新运行单个分片。 -使用 `gh run download` 下载产物。保留原始文件及来源信息; -`scripts/consolidate_results.py` 不属于 CI 流程。 - -## 本地验证 - -生成计划需要 Python 3.11 或更新版本。计算节点控制代码使用现有的 Python 3.10+ -Slurm 主机环境。CPU 测试执行真实规划器、基准编排和启动器,仅替换外部 GPU/Slurm 依赖。 +## 本地测试 ```bash uv run --no-project --python 3.12 --with pytest --with pyyaml --with torch --with numpy \ python -m pytest operatorx/tests/ -q ``` - -实际验收还需要在每个所选运行器上执行带产物的 smoke 运行、失败分片重跑,以及确认释放分配的取消测试。 -CPU 检查不能证明 GPU 兼容性或集群存储可见性。 - -最终覆盖汇总为每个请求分片选择最新产物尝试,保留此前尝试中已成功的分片, -并在任一分片缺失或失败时报告失败。汇总区分请求形状数和结果行数, -因为一个形状可能在多个后端上执行。 - -清理失败后,可在单分片触发中将 `recovery_run_id` 设为同一运行器上近期的 OperatorX -运行 ID。工作流下载执行产物,在申请新节点前重试分配与暂存清理。恢复流程校验运行、 -运行器和私有暂存父目录;不要选择无关或过旧的 Slurm 执行。 - -`cleanup.log` 记录用于确认分配已释放的活动作业查询。查询使用当前用户的作业列表, -因为直接查询已删除的作业 ID,即使分配已终止,也可能返回 Slurm 错误。 - -## AMD 执行 - -`platforms.json` 在 CollectiveX 配置之上按运行器标签补充 AMDS Slurm 集群。 -AMD 接受单卡 `torch`/`vllm` GEMM。ROCm PyTorch 通过 `torch.cuda` 使用 HIP 事件计时; -FP8 在 gfx942 上选择 FNUZ,在 gfx950 上选择 OCP。不支持的格式会明确记录。 -暂存目录由 `RUNNER_TEMP` 推导,位于共享运行器根目录下、`_work` 之外。容器不写入 -源码检出目录。MI300X/MI325X 显式传递 `/dev/kfd` 和 `/dev/dri`;CPU 请求沿用各推理启动器。 - -实验算子变更记录在相邻的 `perf-changelog.yaml`,与根目录中受推理 -配置键约束的变更日志分开维护。 diff --git a/operatorx/CLUSTERS.md b/operatorx/CLUSTERS.md deleted file mode 100644 index 45a8562443..0000000000 --- a/operatorx/CLUSTERS.md +++ /dev/null @@ -1,111 +0,0 @@ -# Clusters - -Reference for the clusters operatorx targets: hardware, how the runner is -launched per platform, and the per-platform quirks. Access details -(hostnames, credentials, checkout paths) are deployment-specific and are not -included here. Fill them in for your own environment. - -Platform → cluster routing lives in `operatorx/clusters.py` (`CLUSTER_PLATFORMS`). -SLURM/env defaults live in `scripts/submit_run.py` (`DEFAULT_CLUSTER`). - -## Conventions - -- `submit_run.py` is **SLURM-only**. It submits `sbatch`/`srun` jobs and is - used on the NVIDIA and AMD clusters. -- Point `OPERATORX_SQUASH_DIR` at your container squash directory, and set - `OPERATORX_PARTITION` / `OPERATORX_ACCOUNT` / `OPERATORX_QOS` to your - cluster's SLURM values (see the env-var table at the bottom). -- Set `OPERATORX_JOB_NAME` to a recognizable SLURM job name for your runs. - -## At a glance - -| Cluster id | Platform | Hardware | Scheduler | -|---------------|----------|---------------------------------|------------| -| `b200_dgx_8x` | nvidia | 8× B200 SXM (DGX-style) | SLURM | -| `b300_hgx_8x` | nvidia | 8× B300 (HGX-style) | SLURM | -| `b200_nvl72` | nvidia | B200 NVL72 (routing only) | SLURM | -| `mi355x_8x` | amd | 8× MI355 OAM per node | SLURM | - -## NVIDIA - -### b200 — DGX-style, 8× B200 SXM (`b200_dgx_8x`) - -| Setting | Value | -|----------|----------------------------------------------------------| -| Hardware | 8× B200 SXM per node | -| GRES | e.g. `gpu:nvidia_b200:8` | -| Defaults | `submit_run.py` defaults target this cluster (`OPERATORX_CLUSTER=b200_dgx_8x`, partition/squash dir from the script/env) | - -```bash -OPERATORX_JOB_NAME= python3 scripts/submit_run.py nvidia -``` - -### b300 — HGX-style, 8× B300 (`b300_hgx_8x`) - -| Setting | Value | -|-------------------|--------------------------------------------------------------| -| Hardware | 8× B300 per node | -| OS note | If the login node runs Python 3.10, `submit_run.py` falls back to `tomli` for TOML parsing | - -Override the SLURM knobs for your cluster: - -```bash -OPERATORX_CLUSTER=b300_hgx_8x \ -OPERATORX_PARTITION= \ -OPERATORX_ACCOUNT= \ -OPERATORX_QOS= \ -OPERATORX_SQUASH_DIR= \ -OPERATORX_BACKENDS=vllm \ -OPERATORX_JOB_NAME= \ -python3 scripts/submit_run.py nvidia -``` - -### `b200_nvl72` - -Present in `CLUSTER_PLATFORMS` (routes to nvidia) as a placeholder. There is no -hardware-specific guidance yet. - -## AMD — `mi355x_8x` - -| Setting | Value | -|-----------|--------------------------------------------------------------| -| Hardware | 8× MI355 OAM per node | -| GRES | e.g. `gpu:amd_instinct_mi355_oam:8` | -| Backends | `containers.toml` `amd.torch` / `amd.vllm` → `vllm-openai-rocm` | - -```bash -OPERATORX_CLUSTER=mi355x_8x \ -OPERATORX_PARTITION= \ -OPERATORX_SQUASH_DIR= \ -OPERATORX_JOB_NAME= \ -python3 scripts/submit_run.py amd -``` - -## Cluster id → platform (`operatorx/clusters.py`) - -```python -CLUSTER_PLATFORMS = { - "b200_dgx_8x": "nvidia", - "b300_hgx_8x": "nvidia", - "b200_nvl72": "nvidia", - "mi355x_8x": "amd", -} -``` - -## `submit_run.py` env vars - -| Var | Default | Notes | -|------------------------|-------------------------------|---------------------------------------------------------------------| -| `OPERATORX_CLUSTER` | per-platform (see script) | Routes the runner via `CLUSTER_PLATFORMS` | -| `OPERATORX_PARTITION` | site default | SLURM partition | -| `OPERATORX_ACCOUNT` | (omitted) | SLURM `--account` | -| `OPERATORX_QOS` | (omitted) | SLURM `--qos` | -| `OPERATORX_SQUASH_DIR` | site default | Where `.sqsh` lives (used by `srun --container-image=`) | -| `OPERATORX_BACKENDS` | all backends for the platform | CSV allowlist | -| `OPERATORX_JOB_NAME` | `benchmark` | SLURM job name | -| `OPERATORX_TESTLISTS` | all under `testlists/` | CSV of testlist stems to run | -| `OPERATORX_TIME_MIN` | `30` | `--time` (minutes) | - -`WORLD_SIZES = [1, 2, 4, 8]` supports single-node runs only. Multi-node NCCL IB bring-up -currently hangs on the B200/B300 fabrics. `MASTER_ADDR` is derived by parsing -`SLURM_NODELIST`. diff --git a/operatorx/README.md b/operatorx/README.md index 2fb365cb0a..c81ff849f4 100644 --- a/operatorx/README.md +++ b/operatorx/README.md @@ -1,75 +1,31 @@ # operatorx -Multi-platform inference operator benchmark suite. Times one op at a time -(gemm, moe, and attention modules) on NVIDIA and AMD and -emits one JSON per run under `results///`. +Inference operator benchmarks: times one op at a time on NVIDIA and AMD through the serving framework's own layers (vLLM). -Attention ops are whole modules, one op type each: `mla`, `mla_dsa`, `dsv4_attn`, -`gqa`, `qsa`, `gdn`, `kda` (`ops/attention.py` lists what each one times). The vLLM -backend builds a one- to six-layer model of the checkpoint family that has the module -(`runners/common/vllm/attention_models.json`) with dummy weights and schedules the op's -batch through vLLM's own model runner, so vLLM picks the attention backend, KV cache -layout and CUDA-graph use as it does when serving. Dispatch `attn_*` testlists on their -own: each builds its own in-process vLLM engine. +| Op type | Schema | What it is | +| --- | --- | --- | +| `gemm` | `ops/gemm.py` | dense GEMM, quantized operands (`a` activation, `b` weight) | +| `moe` | `ops/moe.py` | one MoE layer, router GEMM to combined output | +| `mla`, `mla_dsa`, `dsv4_attn`, `gqa`, `qsa`, `gdn`, `kda` | `ops/attention.py` | whole attention modules, one op type each | -See `CLUSTERS.md` for how to reach each cluster and the per-host quirks. +- Split over devices: each case's `parallel` arg (`{"tp", "dp", "ep", "dcp"}`), defined in `core/parallel.py`. +- Testlists: `testlists/*.json`; every entry lists its `sources` (`//`). Dispatch `attn_*` testlists apart from gemm / moe ones (each builds its own vLLM engine). -## Running on a cluster +## Running -`scripts/submit_run.py ` fans out one sbatch job per -`(container_image, world_size)` pair to your local SLURM. Each job runs -`python -m operatorx` inside the container. - -### b200 (DGX-style, 8x B200 SXM) +Dispatch the **OperatorX Sweep** workflow ([CI.md](CI.md)). The same planner runs by hand on a login node: ```bash -ssh tailscale-b200 -cd /home/sa-shared/harrison/oss-inference-tracker/operatorx -python3 scripts/submit_run.py nvidia +PYTHONPATH=.:inferencex-e2e python3 -m operatorx.ci plan --platform-config operatorx/platforms.json \ + --runner cluster:h200-dgxc --backends vllm --testlists gemm_parallel --world-sizes 1,2,4,8 \ + --chunk-size 500 --mode timing --run-id local --attempt 1 --source-sha "$(git rev-parse HEAD)" \ + --out manifest.json ``` -Defaults that apply: `OPERATORX_CLUSTER=b200_dgx_8x`, -`OPERATORX_PARTITION=gpu-2`, `OPERATORX_SQUASH_DIR=/home/sa-shared/containers`. - -### b300 (HGX-style, 8x B300) - -The b300 cluster needs a non-default partition + account + qos and has its own -squash dir. - -```bash -ssh tailscale-b300 -cd /data/home/sa-shared/harrison/oss-inference-tracker/operatorx -OPERATORX_CLUSTER=b300_hgx_8x \ -OPERATORX_PARTITION=batch_1 \ -OPERATORX_ACCOUNT=benchmark \ -OPERATORX_QOS=batch_1_qos \ -OPERATORX_SQUASH_DIR=/data/home/sa-shared/harrison/containers \ -OPERATORX_BACKENDS=vllm \ -python3 scripts/submit_run.py nvidia -``` - -## Env vars honored by `submit_run.py` - -| Var | Default | Notes | -|-----|---------|-------| -| `OPERATORX_CLUSTER` | per-platform (see script) | Cluster id used to route the runner (see `operatorx.clusters.CLUSTER_PLATFORMS`). | -| `OPERATORX_PARTITION` | `gpu-2` | SLURM partition. | -| `OPERATORX_ACCOUNT` | (omitted) | SLURM `--account`. | -| `OPERATORX_QOS` | (omitted) | SLURM `--qos`. | -| `OPERATORX_SQUASH_DIR` | `/home/sa-shared/containers` | Where `.sqsh` lives. | -| `OPERATORX_BACKENDS` | all backends for the platform | CSV allowlist. | -| `OPERATORX_JOB_NAME` | `benchmark` | SLURM job name. Use `h-benchmark` for benchmark runs (see `CLUSTERS.md`). | - -`WORLD_SIZES` in the script is `[1, 2, 4, 8]` and supports single-node runs only. -Values above 8 are disabled because multi-node NCCL IB bring-up currently hangs on b200/b300. -`MASTER_ADDR` is derived by parsing `SLURM_NODELIST`. +Results: one JSON per run under `results///`; in CI, the `operatorx-shard-*` artifacts, then ingested into the OperatorX database. ## Adding a backend / op -- Backend impl: `operatorx/runners//backends/.py`, exports an - `IMPLS = [BackendImpl(...)]` list. -- Op spec: `operatorx/ops/.py`, calls `register(OpSpec(..., flops=, bytes=))`. -- Add the container image to `containers.toml`, then on each cluster run - `python3 scripts/pull_containers.py nvidia` to enroot-import it into - `$OPERATORX_SQUASH_DIR`. -- Add shapes to `testlists/.json`. +- Backend: `runners//backends/.py` exporting `IMPLS = [BackendImpl(...)]`; its image in `containers.toml`. +- Op: `ops/.py` calling `register(OpSpec(type=..., arg_schema=..., parallel_axes=...))`. +- Shapes: `testlists/.json`. diff --git a/operatorx/ci.py b/operatorx/ci.py index b5f075748c..921fdef86e 100644 --- a/operatorx/ci.py +++ b/operatorx/ci.py @@ -20,9 +20,8 @@ from pathlib import Path ROOT = Path(__file__).resolve().parent -# Each shard runs on the self-hosted runners that carry this label - the labels -# InferenceX's benchmark workflows request - mapped to its Slurm cluster id and the -# CollectiveX platform profile (configs/platform_config.json) its Slurm settings come from. +# runner label (as InferenceX's workflows request) -> (Slurm cluster id, CollectiveX +# platform profile in configs/platform_config.json) RUNNERS = { "cluster:h100-dgxc": ("h100_dgxc_8x", "h100-dgxc"), "cluster:h200-dgxc": ("h200_dgxc_8x", "h200-dgxc"), @@ -35,7 +34,7 @@ "cluster:mi355x-amds": ("mi355x_8x", "mi355x"), } MODES = ("timing", "counters") -# Hardware counters per kernel. Latencies from a counters run are perturbed by the profiler. +# per-kernel hardware counters; a counters run's latencies are perturbed by the profiler NCU_METRICS = ",".join(( "gpu__time_duration.sum", "sm__cycles_elapsed.avg.per_second", "gpc__cycles_elapsed.avg.per_second", "dram__cycles_elapsed.avg.per_second", @@ -58,7 +57,7 @@ "l1tex__data_pipe_lsu_wavefronts_mem_shared_op_ld.sum", "l1tex__data_pipe_lsu_wavefronts_mem_shared_op_st.sum", )) -# Host Nsight Compute: real installs (/nsight-compute[-]/ncu), then any ncu on PATH. +# host Nsight Compute: real installs (/nsight-compute[-]/ncu), then PATH NCU_SEARCH = ( "ls -d /opt/nvidia/nsight-compute/*/ncu /usr/local/cuda*/nsight-compute*/ncu " "/opt/nvidia/nsight-compute*/ncu $(command -v ncu) 2>/dev/null | xargs -r readlink -f | sort -u" @@ -74,7 +73,7 @@ def version(p: Path) -> tuple: if real: return max(real, key=version) return Path(paths[-1]) if paths else None -# rocprofv3 fits only a few counters per hardware pass, so a counters run repeats per pass. +# rocprofv3 fits only a few counters per hardware pass: a counters run repeats per pass ROCPROF_PASSES = { "fetch": ["FETCH_SIZE"], "hm": ["TCC_HIT_sum", "TCC_MISS_sum"], @@ -98,8 +97,7 @@ def version(p: Path) -> tuple: def load_platforms(path: Path) -> dict: - """Hardware and Slurm settings per runner label: the CollectiveX profile the label - maps to, overlaid with this file's own entry for the label.""" + """Per runner label: its CollectiveX profile overlaid with this file's entry.""" document = json.loads(path.read_text()) base = {} if "base" in document: @@ -112,6 +110,33 @@ def load_platforms(path: Path) -> dict: return platforms +# CollectiveX's network profile scrub (runtime/common.sh collx_apply_network_profile). +_NETWORK_ENV = ("NCCL_NET", "NCCL_SOCKET_IFNAME", "NCCL_IB_", "NVSHMEM_", "MORI_RDMA_", "UCCL_") + + +def stage_model_config(checkpoint: str, root: Path) -> None: + """The checkpoint's config.json, without weights.""" + import urllib.request + + dest = root / checkpoint / "config.json" + dest.parent.mkdir(parents=True, exist_ok=True) + try: + with urllib.request.urlopen(f"https://huggingface.co/{checkpoint}/raw/main/config.json", + timeout=60) as r: + dest.write_bytes(r.read()) + except OSError as e: + print(f"[operatorx] no config.json for {checkpoint}: {e}", flush=True) + + +def _split_axes(op_type: str) -> tuple[str, ...]: + import operatorx.ops # noqa: F401 registers the op specs + from operatorx.core import op_registry, parallel + try: + return op_registry.get(op_type).parallel_axes or ("tp",) + except KeyError: + return parallel.AXES + + def family(name: str) -> str: """Hardware family of a runner label: 'cluster:mi355x-amds' -> 'mi355x'.""" return name.removeprefix("cluster:").split("-")[0] @@ -139,9 +164,10 @@ def plan( mode: str = "timing", find_recipe=None, ) -> dict: - """Shards for a selection. A case whose source checkpoint InferenceX serves on this - hardware runs under that recipe (find_recipe(framework, hardware, checkpoints), - operatorx.recipes.find): its image, launch env and server arguments.""" + """Shards for a selection, one per (parallel split, InferenceX recipe). A case whose + source checkpoint InferenceX serves on this hardware runs under that recipe + (find_recipe = operatorx.recipes.find): its image, launch env and server arguments.""" + from operatorx.core import parallel from operatorx.recipes import FRAMEWORKS, checkpoints if runner not in RUNNERS: raise ValueError(f"unsupported runner: {runner}") @@ -158,20 +184,19 @@ def plan( raise ValueError("select at least one registered backend for this GPU platform") if is_amd(runner) and ( set(backends) - {"torch", "vllm"} - or world_sizes != [1] or any( shape["type"] not in {"gemm", "moe", "mla", "mla_dsa", "dsv4_attn", "gqa", "qsa", "gdn", "kda"} for shapes in testlists.values() for shape in shapes ) ): - raise ValueError( - "AMD CI supports single-GPU torch/vllm GEMM" - ) + raise ValueError("AMD CI supports the torch/vllm GEMM, MoE and attention ops") if not world_sizes or set(world_sizes) - {1, 2, 4, 8}: raise ValueError("world sizes must be selected from 1,2,4,8 (single node)") if any(ws > gpus for ws in world_sizes): raise ValueError(f"world size exceeds the runner's {gpus}-GPU physical node") + if "torch" in backends and world_sizes != [1]: + raise ValueError("the torch backend runs single-device GEMM only") if not 1 <= chunk_size <= 500: raise ValueError("chunk size must be between 1 and 500") groups = defaultdict(list) @@ -179,35 +204,38 @@ def plan( excluded = 0 for name, shapes in sorted(testlists.items()): for shape in shapes: - args = shape["args"] - ws = args.get("world_size", 1) - if type(ws) is not int or ws < 1: - raise ValueError("world_size must be a positive integer") - if ws not in world_sizes: + split = parallel.normalize(shape["args"].get("parallel")) + if parallel.world_size(split) not in world_sizes: excluded += 1 continue recipe = next((r for b in backends if b in FRAMEWORKS and find_recipe - for r in [find_recipe(b, family(runner), checkpoints(shape.get("sources")))] if r), - None) + for r in [find_recipe(b, family(runner), checkpoints(shape.get("sources")), + split, _split_axes(shape["type"]))] if r), None) if recipe: recipes_seen[recipe["recipe"]] = recipe - groups[(ws, recipe["recipe"] if recipe else "")].append({"testlist": name, "shape": shape}) + # attention builds its own engine; never in a process with gemm/moe + state = "layer" if shape["type"] in ("gemm", "moe") else "engine" + groups[(json.dumps(split, sort_keys=True), recipe["recipe"] if recipe else "", state)].append( + {"testlist": name, "shape": shape}) image_groups = defaultdict(list) for backend in sorted(set(backends)): image_groups[images[backend]["image"]].append(backend) cells = [] shards = [] - for (ws, recipe), cases in sorted(groups.items()): + for (split, recipe, _state), cases in sorted(groups.items()): + split = json.loads(split) + ws = parallel.world_size(split) rest = image_groups if recipe: r = recipes_seen[recipe] framework = next(b for b in backends if b in FRAMEWORKS) shards.append((r["image"], [framework], ws, cases, - {"recipe": recipe, "checkpoint": r["checkpoint"], "env": r["env"], - "engine_args": r["engine_args"]})) + {"parallel": split, "recipe": recipe, "recipe_match": r["match"], + "checkpoint": r["checkpoint"], "env": r["env"], "engine_args": r["engine_args"]})) # the other backends run the same cases on their own images rest = {i: [b for b in bs if b != framework] for i, bs in image_groups.items()} - shards += [(image, selected, ws, cases, {}) for image, selected in sorted(rest.items()) if selected] + shards += [(image, selected, ws, cases, {"parallel": split}) + for image, selected in sorted(rest.items()) if selected] for image, selected, ws, cases, extra in shards: for offset in range(0, len(cases), chunk_size): cell = { @@ -304,13 +332,13 @@ def cleanup(root: Path, timeout_seconds: int) -> None: def image_key(image: str, digest: str, image_platform: str) -> str: - # Preserve the already-qualified amd64 cache while isolating Arm imports. + # keep the existing amd64 cache keys; Arm imports get their own suffix = "" if image_platform == "linux/amd64" else f":{image_platform}" return hashlib.sha256((image + digest + suffix).encode()).hexdigest() def shared_base(profile: dict, runner: str) -> Path: - """Resolve the runner's configured/shared account storage, never temporary HOME.""" + """The runner's configured/shared account storage, never a temporary HOME.""" if profile.get("stage_dir"): roots = [Path(profile["stage_dir"])] elif is_amd(runner): @@ -322,8 +350,8 @@ def shared_base(profile: dict, runner: str) -> Path: raise ValueError("AMD staging requires the shared runner _work/_temp path") roots = [runner_temp.parent.parent] elif family(runner) == "b300": - # CollectiveX uses the compute-visible account home on these nodes. - # The shared squash parent is not writable by the GHA service account. + # compute-visible account home, as CollectiveX; the shared squash parent isn't + # writable by the GHA service account roots = [Path(pwd.getpwuid(os.getuid()).pw_dir)] elif profile.get("squash_dir"): roots = [Path(profile["squash_dir"]).parent] @@ -336,7 +364,7 @@ def shared_base(profile: dict, runner: str) -> Path: def import_image(args) -> None: - # Runs on the configured import host with a compute-visible cache and lock. + # runs on the import host, with a compute-visible cache and lock import fcntl image, digest = args.image, args.digest @@ -363,10 +391,9 @@ def import_image(args) -> None: try: host, repository, tag = probe_module().registry_reference(image) uri = f"docker://{host}#{repository}:{tag}" - # B300 login/compute homes are node-local; every importer gets private - # temporary paths. Preserve an explicitly configured shared cache. Where - # /tmp cannot hold overlay whiteouts (H100 DGXC) the node's own enroot - # paths are kept, as its InferenceX launcher imports with them. + # B300 homes are node-local: private temp paths unless a shared cache is + # configured; where /tmp can't hold overlay whiteouts (H100 DGXC) keep the + # node's enroot paths, as its InferenceX launcher does with tempfile.TemporaryDirectory(prefix="operatorx-enroot-") as scratch: env = dict(os.environ) private = os.environ.get("OPERATORX_ENROOT_DEFAULTS") != "1" @@ -387,8 +414,7 @@ def import_image(args) -> None: env=env, ) subprocess.run(["unsquashfs", "-s", str(temporary)], check=True) - # The importer uses the tag, just like CollectiveX. Refuse a tag that moved - # between planning and import rather than mislabelling the measurement. + # imported by tag (as CollectiveX); refuse a tag that moved since planning if probe_module().resolve_image_digest(image) != digest: raise RuntimeError( "image tag moved or digest verification failed; dispatch again" @@ -493,10 +519,14 @@ def interrupted(signum, frame): stage / "source/operatorx", ignore=shutil.ignore_patterns("__pycache__", "results", ".venv", "tests"), ) - shutil.copytree( - ROOT.parent / "collectivex/runtime", - stage / "source/collectivex/runtime", - ) + for part in ("runtime", "bench"): # probes; the EP harness's cross-rank timing + shutil.copytree( + ROOT.parent / "collectivex" / part, + stage / "source/collectivex" / part, + ignore=shutil.ignore_patterns("__pycache__"), + ) + if cell.get("checkpoint"): + stage_model_config(cell["checkpoint"], stage / "models") for name in sorted({c["testlist"] for c in cell["cases"]}): write_json( stage / "testlists" / f"{name}.json", @@ -553,8 +583,7 @@ def interrupted(signum, frame): "--image-platform", image_platform, ] - # Import on the allocated architecture, including B300: its submit host - # lacks PyTorch extraction space, as the inference launcher notes. + # import on the allocated node (B300's submit host lacks extraction space) import_env = dict(os.environ) if profile.get("enroot_cache_path"): import_env["ENROOT_CACHE_PATH"] = profile["enroot_cache_path"] @@ -583,9 +612,17 @@ def interrupted(signum, frame): PYTHONPATH="/opx/source", PYTHONDONTWRITEBYTECODE="1", WORLD_SIZE=str(cell["world_size"]), + OPERATORX_PARALLEL=json.dumps(cell.get("parallel") or {}), MASTER_ADDR="127.0.0.1", - MASTER_PORT="29500", + MASTER_PORT=str(29500 + int(hashlib.sha256(f"{job}:{cell['id']}".encode()).hexdigest(), 16) % 2000), ) + if (stage / "models" / cell.get("checkpoint", "-") / "config.json").is_file(): + env["OPERATORX_MODEL_CONFIG"] = f"/opx/models/{cell['checkpoint']}" + # --export=ALL forwards the runner's network profile; single node needs none + for key in [k for k in env if k.startswith(_NETWORK_ENV)]: + del env[key] + if not is_amd(cell["runner"]): + env["NCCL_CUMEM_ENABLE"] = "1" env.update(cell.get("env", {})) # the InferenceX recipe's launch env mounts = f"{stage}:/opx" # NVIDIA images carry no Nsight Compute; counters runs mount the node's newest. @@ -597,8 +634,7 @@ def interrupted(signum, frame): (root / "ncu.log").write_text("\n".join(found) + "\n") ncu = pick_ncu(found) if ncu is not None: - # ncu is a wrapper that finds its install next to itself (../), so mount the - # whole install tree at the same path + # ncu finds its install relative to itself: mount the whole tree in place mounts += f",{ncu.parent.parent}:{ncu.parent.parent}" env["OPERATORX_NCU"] = str(ncu) if hw in ("mi300x", "mi325x"): @@ -623,13 +659,14 @@ def interrupted(signum, frame): run.append("--container-remap-root") if hw == "b300": run.append("--mpi=none") - # The Python entrypoint preserves the allocated GPU mask without a shell. Run it - # by path: an image's own PYTHONPATH (ROCm images set one) replaces the host's. + # by path: an image's own PYTHONPATH (ROCm images set one) replaces the host's run += ["python3", "/opx/source/operatorx/ci.py", "rank"] + # bound a collective that hangs without failing + run = ["timeout", "-k", "30", str(max(300, args.time_minutes * 60 - 300)), *run] command(run, root / "benchmark.log", env=env) rc = 0 finally: - # Stop writers before collecting; failed cleanup retains the evidence. + # stop writers before collecting; a failed cleanup keeps the evidence for sig in (signal.SIGINT, signal.SIGTERM, signal.SIGHUP): signal.signal(sig, signal.SIG_IGN) finalize(root, args.cleanup_seconds) @@ -668,7 +705,7 @@ def rank() -> None: ] if os.environ.get("OPERATORX_MODE", "timing") != "counters": os.execv(sys.executable, bench) - # One profiled replay per op, marked so counters join to ops; no timing warmups. + # one marked, profiled replay per op so counters join to ops; no timing warmups os.environ.update( OPERATORX_PROFILE="1", OPERATORX_PROFILE_MARKERS="1", @@ -811,7 +848,7 @@ def main() -> None: } vendor = "amd" if is_amd(args.runner) else "nvidia" images = tomllib.loads((ROOT / "containers.toml").read_text())[vendor] - # Fail here, before any node is allocated, for a backend with no module. + # fail before any node is allocated for a backend with no module missing = [ b for b in args.backends.split(",") if not (ROOT / "runners" / vendor / "backends" / f"{b}.py").is_file() @@ -833,8 +870,8 @@ def main() -> None: for image in digests: digests[image] = probe_module().resolve_image_digest(image) for cell in result["include"]: - # a recipe whose image tag the registry no longer serves (pruned nightlies) runs on - # the backend's own image with the recipe's env; the manifest records the swap + # recipe image tag gone from the registry (pruned nightly): backend image + recipe + # env; the manifest records the swap if not digests[cell["image"]] and cell.get("recipe"): cell["recipe_image_unavailable"] = cell["image"] cell["image"] = images[cell["backends"][0]]["image"] diff --git a/operatorx/clusters.py b/operatorx/clusters.py index 29e2c8771b..17d87c2b86 100644 --- a/operatorx/clusters.py +++ b/operatorx/clusters.py @@ -1,8 +1,4 @@ -"""Minimal cluster routing table. - -Maps cluster id -> platform (for runner dispatch) and cluster id -> chip -(for legacy/grouped layouts). No peak-throughput or bandwidth info. -""" +"""Cluster id -> platform (runner dispatch) and cluster id -> chip.""" from __future__ import annotations diff --git a/operatorx/core/backend.py b/operatorx/core/backend.py index ee43453ee9..452b2b82d2 100644 --- a/operatorx/core/backend.py +++ b/operatorx/core/backend.py @@ -11,9 +11,8 @@ class BackendImpl: """prepare(op) -> ctx; kernel(ctx) runs the op once. - launcher(ctx), when given, returns (callable, cuda_graph): what the runner times - and profiles instead of kernel(ctx) - how the backend's framework would execute - the op - and whether that is a CUDA-graph replay.""" + launcher(ctx), if set, returns (callable, cuda_graph): what the runner times instead of + kernel(ctx) (how the framework would execute the op) and whether it is a graph replay.""" op_type: str prepare: Callable[[Op], Any] kernel: Callable[[Any], None] @@ -21,15 +20,8 @@ class BackendImpl: def lookup_versions(*pkg_names: str) -> dict[str, str]: - """Resolve installed versions for a set of package names. - - Tries ``importlib.metadata.version`` first (catches pip-installed - packages), then falls back to importing the module and reading - ``__version__`` (catches PYTHONPATH-installed checkouts like MaxText). - Missing packages are silently skipped. - - Used by each backend's ``versions()`` to declare what it depends on. - """ + """Installed versions: package metadata, else the module's __version__ (PYTHONPATH + checkouts); missing packages are skipped.""" import importlib as _importlib out: dict[str, str] = {} diff --git a/operatorx/core/errors.py b/operatorx/core/errors.py index 4aac89a7a6..cce52ce653 100644 --- a/operatorx/core/errors.py +++ b/operatorx/core/errors.py @@ -1,7 +1,2 @@ class UnsupportedOpError(Exception): - """Backend cannot execute this op (wrong dtype, wrong category, kernel missing, ...). - - Distinct from a runtime crash: the backend simply doesn't claim support for this - (op_type, args) combination. Smoke tests record this as ``status="unsupported"`` - rather than ``"error"``. - """ + """The backend doesn't support this (op_type, args); recorded as status="unsupported", not "error".""" diff --git a/operatorx/core/op.py b/operatorx/core/op.py index 9f4eeae135..48f32f82ff 100644 --- a/operatorx/core/op.py +++ b/operatorx/core/op.py @@ -11,11 +11,8 @@ class Op: type: str args: Mapping[str, Any] backend: str - # Where the case comes from: one "/" per layer that runs this op, - # e.g. "deepseek-ai/DeepSeek-V4-Pro/q_a_proj" (a checkpoint id is "org/model", so the - # role is what follows the last "/"). A shape shared by several models, or by several - # roles in one, lists every pair; a shape from no model lists none. Equality/hash - # ignore it - the same (type, args, backend) is the same op whichever model it came from. + # one "//" per layer that runs this op (e.g. + # "deepseek-ai/DeepSeek-V4-Pro/q_a_proj"); ignored by equality/hash sources: tuple[str, ...] = () def __post_init__(self) -> None: @@ -41,3 +38,5 @@ class OpSpec: type: str arg_schema: type description: str = "" + # axes (core/parallel.py) the "parallel" arg may use; empty: one device only + parallel_axes: tuple[str, ...] = () diff --git a/operatorx/core/op_registry.py b/operatorx/core/op_registry.py index ccf21915bb..73751c4e42 100644 --- a/operatorx/core/op_registry.py +++ b/operatorx/core/op_registry.py @@ -2,6 +2,7 @@ from typing import Sequence +from operatorx.core import parallel from operatorx.core.op import Op, OpSpec _REGISTRY: dict[str, OpSpec] = {} @@ -24,9 +25,12 @@ def all_ops() -> Sequence[OpSpec]: def validate(op: Op) -> Op: + """Check args against the schema; "parallel" is checked against the op type's axes.""" spec = get(op.type) + args = dict(op.args) + parallel.check(args.pop("parallel", None), spec.parallel_axes) try: - spec.arg_schema(**dict(op.args)) + spec.arg_schema(**args) except TypeError as e: raise ValueError(f"op {op.type!r} args don't match schema: {e}") from e return op diff --git a/operatorx/core/parallel.py b/operatorx/core/parallel.py new file mode 100644 index 0000000000..7fe37ad787 --- /dev/null +++ b/operatorx/core/parallel.py @@ -0,0 +1,44 @@ +"""How an op is split over the devices of one node. + + "parallel": {"tp": T, "dp": D, "ep": E, "dcp": C} (each optional, default 1) + + tp: weights (attention: heads) split T ways; the reduction is part of the op. + dp: D groups each bring their own batch (MoE experts see D x tokens in total). + ep: whole experts partitioned E ways; E is 1 or T x D. + dcp: each request's KV cache split C ways within a tp group, outputs merged; C divides T. + +Runs on T x D devices. Args keep the full unsplit shape and each dp group's batch; how +the split is carried out is the backend's choice. No "parallel": one device. +""" +from __future__ import annotations + +from typing import Any + +AXES = ("tp", "dp", "ep", "dcp") +GEMM_AXES = ("tp",) +MOE_AXES = ("tp", "dp", "ep") +ATTENTION_AXES = ("tp", "dp", "dcp") + + +def check(p: Any, allowed: tuple[str, ...] = AXES) -> None: + if p is None: + return + if not isinstance(p, dict) or set(p) - set(allowed): + raise ValueError(f"parallel must be a dict with keys from {list(allowed)}, got {p!r}") + for k, v in p.items(): + if not isinstance(v, int) or isinstance(v, bool) or v < 1: + raise ValueError(f"parallel.{k} must be a positive int, got {v!r}") + if p.get("ep", 1) not in (1, world_size(p)): + raise ValueError("parallel.ep must be 1 or tp x dp") + if p.get("tp", 1) % p.get("dcp", 1): + raise ValueError("parallel.dcp must divide parallel.tp") + + +def world_size(p: dict | None) -> int: + p = p or {} + return p.get("tp", 1) * p.get("dp", 1) + + +def normalize(p: dict | None) -> dict: + """Every axis spelled out: the key a process's parallel state is built for.""" + return {k: (p or {}).get(k, 1) for k in AXES} diff --git a/operatorx/core/result.py b/operatorx/core/result.py index 0fa23d1c25..160493132c 100644 --- a/operatorx/core/result.py +++ b/operatorx/core/result.py @@ -17,8 +17,8 @@ class Result: op: Op metrics: Mapping[str, float] = field(default_factory=dict) status: str = "ok" # "ok" | "unsupported" | "error" - message: str | None = None # only set when status != "ok" - testlist: str | None = None # which testlist file produced this op + message: str | None = None # only when status != "ok" + testlist: str | None = None # testlist file that produced this op def to_dict(r: Result) -> dict: @@ -57,11 +57,7 @@ def _result_from_dict(d: dict) -> Result: def write_run_result(path: Path | str, run: RunInfo, results: Iterable[Result]) -> None: - """Write one run's worth of results to ``path`` in the run-result wrapper shape. - - Body: ``{"schema_version", "run": {...RunInfo...}, "rows": [{op, metrics}, ...]}``. - Parent directories are created. Existing files are overwritten — one file = one run. - """ + """{"schema_version", "run", "rows"}; one file = one run, replaced atomically.""" path = Path(path) path.parent.mkdir(parents=True, exist_ok=True) body = { @@ -75,7 +71,6 @@ def write_run_result(path: Path | str, run: RunInfo, results: Iterable[Result]) def read_run_result(path: Path | str) -> tuple[RunInfo, list[Result]]: - """Inverse of :func:`write_run_result`. Returns ``(run, rows)``.""" path = Path(path) body = json.loads(path.read_text()) run = _run_from_dict(body["run"]) diff --git a/operatorx/main.py b/operatorx/main.py index 8938306ba5..0caaf632d9 100644 --- a/operatorx/main.py +++ b/operatorx/main.py @@ -1,19 +1,8 @@ -"""Run every testlist × every backend on the detected platform. +"""``python -m operatorx``: run every testlist x backend on this platform. -Invoke as ``python -m operatorx`` (or ``python -m operatorx.main``). - -Reads testlists from ``/testlists/*.json`` (one file per testlist; filename -without extension = testlist name). Default = all testlists; use ``--testlists a,b,c`` -to filter. - -Backend op-type support is derived from each backend module's ``IMPLS`` list. If a -backend doesn't claim an op_type, the (op, backend) combination is silently skipped -— no Result row is emitted. Args-level rejection (e.g. dtype mismatch) is captured -as ``status="unsupported"``. Other exceptions become ``status="error"``. - -Platform is detected from ``$OPERATORX_CLUSTER`` → ``CLUSTERS[id].platform``, or -overridden with ``--platform``. Backend filter via ``--backends`` (CSV) or env -``OPERATORX_BACKENDS``. World size from ``WORLD_SIZE`` env (set by torchrun). +Op types a backend's IMPLS doesn't claim are skipped (no row, unless --strict); +UnsupportedOpError -> status="unsupported", other exceptions -> "error". +Env: OPERATORX_CLUSTER, OPERATORX_BACKENDS, OPERATORX_PARALLEL (JSON), WORLD_SIZE/RANK. """ from __future__ import annotations @@ -27,16 +16,15 @@ from pathlib import Path import operatorx.ops # noqa: F401 populates op registry -from operatorx.core import op_registry +from operatorx.core import op_registry, parallel from operatorx import Op, Result, UnsupportedOpError, write_run_result from operatorx.clusters import CLUSTER_PLATFORMS from operatorx.runtime import runtime_snapshot, utc_now_iso -# a testlist source: the checkpoint id ("org/model") and the op's role in it +# a testlist source: "//" _SOURCE = re.compile(r"[^/\s]+/[^/\s]+/[^/\s]+") -# Package directory contains the checked-in testlists and local results. _REPO_ROOT = Path(__file__).resolve().parent @@ -45,9 +33,6 @@ def _discover_backends(platform: str) -> list[str]: - """All backends for a platform = the python modules under - ``operatorx/runners//backends/`` - """ try: pkg = importlib.import_module(f"operatorx.runners.{platform}.backends") except ImportError: @@ -65,7 +50,7 @@ def _csv(s: str | None) -> list[str]: def _load_testlists(names: list[str] | None, directory: Path = TESTLIST_DIR) -> dict[str, list[dict]]: - """Returns {testlist_name: [shape_dict, ...]}. Default = all available.""" + """{testlist_name: [shape_dict, ...]}; all available by default.""" available = {p.stem: p for p in sorted(directory.glob("*.json"))} if names: wanted = {n: available[n] for n in names if n in available} @@ -87,7 +72,6 @@ def _load_testlists(names: list[str] | None, directory: Path = TESTLIST_DIR) -> def _backend_supported_ops(platform: str, backends: list[str], strict: bool = False) -> dict[str, set[str]]: - """Discover per-backend supported op_types from each module's IMPLS list.""" out: dict[str, set[str]] = {} for b in backends: try: @@ -103,11 +87,7 @@ def _backend_supported_ops(platform: str, backends: list[str], strict: bool = Fa def _collect_backend_versions(platform: str, backends: list[str]) -> dict[str, str]: - """Ask each backend module to report its own library versions. - - Each backend exports ``versions() -> dict[str, str]``. Returned keys are - merged into ``RunInfo.software``. Missing or failing probes are skipped. - """ + """Each backend's versions(); failing probes are skipped.""" out: dict[str, str] = {} for b in backends: try: @@ -136,11 +116,13 @@ def _resolve_platform(args_platform: str | None, run_cluster: str | None) -> str ) -def _resolve_world_size(platform: str) -> int: - ws = os.environ.get("WORLD_SIZE") - if ws: - return int(ws) - return 1 +def _resolve_split() -> dict: + """This process's split (OPERATORX_PARALLEL, JSON; default one device); must match WORLD_SIZE.""" + split = parallel.normalize(json.loads(os.environ.get("OPERATORX_PARALLEL") or "null")) + ws = int(os.environ.get("WORLD_SIZE", "1")) + if parallel.world_size(split) != ws: + raise SystemExit(f"OPERATORX_PARALLEL={split} needs {parallel.world_size(split)} ranks, launched {ws}") + return split def main() -> int: @@ -159,44 +141,33 @@ def main() -> int: run = runtime_snapshot() platform = _resolve_platform(args.platform, run.cluster) - # Backends requested = _csv(args.backends) or _csv(os.environ.get("OPERATORX_BACKENDS")) backends = requested if requested else _discover_backends(platform) if not backends: raise SystemExit(f"no backends configured for platform={platform!r}") backend_ops = _backend_supported_ops(platform, backends, strict=args.strict) - # Merge per-backend library versions into the run's software dict. run.software.update(_collect_backend_versions(platform, backends)) - ws = _resolve_world_size(platform) + split = _resolve_split() + ws = parallel.world_size(split) rank = int(os.environ.get("RANK", "0")) - # Testlists testlists = _load_testlists( _csv(args.testlists) or _csv(os.environ.get("OPERATORX_TESTLISTS")) or None, args.testlist_dir, ) - # Load runner runner_mod = importlib.import_module(f"operatorx.runners.{platform}.runner") - # Build entries: (op, testlist_name). entries: list[tuple[Op, str]] = [] for tl_name, shapes in testlists.items(): for shape in shapes: - # In a multi-rank job (ws>1) only run ops explicitly tagged with the - # matching world_size — running single-rank ops on every rank is - # both wasteful and meaningless (each rank would redo the same work - # in parallel). Per-rank ops belong in the ws=1 job. - shape_ws = shape["args"].get("world_size") - if shape_ws is None: - if ws != 1: - continue - elif int(shape_ws) != ws: + # one parallel state per process; other splits run in their own process + if parallel.normalize(shape["args"].get("parallel")) != split: continue for backend in backends: if not args.strict and shape["type"] not in backend_ops.get(backend, set()): - continue # Strict CI retains unsupported backend/operator pairs. + continue # strict CI keeps unsupported backend/op pairs entries.append(( Op(type=shape["type"], args=shape["args"], backend=backend, sources=shape["sources"]), @@ -204,7 +175,7 @@ def main() -> int: )) if rank == 0: - print(f"[run_smoke] platform={platform} cluster={run.cluster!r} ws={ws}") + print(f"[run_smoke] platform={platform} cluster={run.cluster!r} parallel={split}") print(f"[run_smoke] backends={backends}") print(f"[run_smoke] testlists={list(testlists)} -> {len(entries)} (op,backend) entries") @@ -242,7 +213,7 @@ def main() -> int: latency_str = f" ERROR ({type(e).__name__})" wall_s = _time.perf_counter() - _t0 if rank == 0: - # Checkpoint outside the timed kernel so cancellation preserves completed rows. + # checkpoint after each op, outside timing, so cancellation keeps completed rows if args.strict: run.finished_at = utc_now_iso() write_run_result(out_path, run, results) @@ -252,10 +223,7 @@ def main() -> int: ) print(f"[ws={ws}] {tl:12} {status:11} {op.type:18} {op.backend:10} " f"{latency_str} wall={wall_s:6.1f}s {shape_str}", flush=True) - # Release per-op tensors back to the CUDA driver so the next op's - # allocator sees the full GPU. Without this, PyTorch's caching - # allocator hangs on to large output buffers and the next big shape - # OOMs even though the prior tensors are no longer referenced. + # release cached blocks, else the next big shape can OOM on the previous op's buffers try: import torch as _torch if _torch.cuda.is_available(): @@ -269,7 +237,6 @@ def main() -> int: write_run_result(out_path, run, results) print(f"\n[run_smoke] {counts} -> {out_path}") - # Multi-rank cleanup (torch.distributed) if ws > 1: try: import torch.distributed as dist diff --git a/operatorx/ops/attention.py b/operatorx/ops/attention.py index de13be1659..d42a24c302 100644 --- a/operatorx/ops/attention.py +++ b/operatorx/ops/attention.py @@ -1,30 +1,24 @@ """Attention modules: one op type per module, each timed whole. -Every op starts at the post-input_layernorm hidden states x [T, hidden] (plus -positions) and ends at the module output [T, hidden] before the residual add. It -includes the module's projections, norms, RoPE, cache/state writes, selection -(indexer, compressor) and the attention itself; the tensor-parallel all-reduce, -residual, hyper-connections and MTP layers are outside it. The KV cache (or linear -attention state) holds each request's ctx tokens of random, format-valid data before -the timed call. +From post-input_layernorm x [T, hidden] (+ positions) to the module output before the +residual add: projections, norms, RoPE, cache/state writes, selection and attention; the +tp all-reduce, residual, hyper-connections and MTP are outside. The cache/state holds +each request's ctx tokens of random, format-valid data. Shared args: - proj: {"": {"a": operand, "b": operand}}. Operand descriptors are - the gemm op's; a projection left out is bf16 x bf16. + proj: {"": {"a": operand, "b": operand}} (gemm operands); omitted: bf16 x bf16. batch: {"groups": [{"count", "q", "ctx"}], "pages"?: "contiguous" | "shuffled", "seed"?} - count requests with q new tokens each and ctx tokens already cached; ctx is an int - or {"dist": "uniform", "min", "max"} drawn per request from seed. Whether a request - runs the backend's prefill or decode path is the backend's choice, recorded with - the result. - selection (sparse modules): which tokens the selection step picks. natural: whatever - the module's indexer computes on the random cache; uniform / recent / clustered - force the indices. + count requests of q new tokens on ctx cached; ctx is an int or {"dist": "uniform", "min", + "max"} drawn from seed. Prefill vs decode path is the backend's choice, recorded. + selection (sparse modules): natural = the module's indexer on the random cache; + uniform / recent / clustered force the indices. """ from __future__ import annotations -from dataclasses import dataclass +from dataclasses import dataclass, replace from typing import Any +from operatorx.core import parallel from operatorx.core.op import OpSpec from operatorx.core.op_registry import register from operatorx.ops.gemm import check_operand @@ -231,12 +225,11 @@ class Dsv4AttnArgs: cache write -> sparse MQA over window + compressed tokens with attn_sink -> inverse RoPE -> wo_a (o_groups grouped) -> wo_b. - compress_ratio: 4 / 128 V4-Pro (4 with an indexer, 128 attends every compressed - token); 0 (window only), 1 / 2 V4.1. The compressor's overlap, positional embedding and - gate follow from the ratio as in the checkpoints. source false: a layer that reads - a pre-filled compressed cache written by another layer and has no compressor. - rope_theta applies to window-only layers, compress_rope_theta (+ rope_scaling) to - compressed ones. index_cache_dtype: the indexer K cache (vLLM indexer_kv_dtype). + compress_ratio: 4 / 128 V4-Pro (4 with an indexer, 128 attends every compressed token); + 0 (window only), 1 / 2 V4.1; compressor overlap/pos-emb/gate follow the ratio. + source false: reads another layer's pre-filled compressed cache, no compressor. + rope_theta: window-only layers; compress_rope_theta (+ rope_scaling): compressed ones. + index_cache_dtype: the indexer K cache (vLLM indexer_kv_dtype). """ hidden: int heads: int @@ -451,4 +444,4 @@ def __post_init__(self): OpSpec(type="gdn", arg_schema=GdnArgs, description="Gated DeltaNet linear attention"), OpSpec(type="kda", arg_schema=KdaArgs, description="Kimi Delta Attention"), ): - register(_spec) + register(replace(_spec, parallel_axes=parallel.ATTENTION_AXES)) diff --git a/operatorx/ops/gemm.py b/operatorx/ops/gemm.py index 8b2e494a3c..afd10b7234 100644 --- a/operatorx/ops/gemm.py +++ b/operatorx/ops/gemm.py @@ -3,6 +3,7 @@ from dataclasses import dataclass from typing import Any +from operatorx.core import parallel from operatorx.core.op import OpSpec from operatorx.core.op_registry import register @@ -68,20 +69,17 @@ def plain_dtype(args: dict) -> str | None: class GemmArgs: """C[M,N] = activation(A[M,K] @ B[N,K]^T + bias); A = activation, B = weight. - a, b: operand descriptors, {"dtype", "scale"?, "scale2"?, "symmetric"?} + a, b: {"dtype", "scale"?, "scale2"?, "symmetric"?, "input"?}; unquantized: {"dtype": "bf16"} dtype: storage element type (bf16, e4m3, e2m1, int4, ...) - scale: {"dtype", "static", "group": [rows, cols]}, the checkpoint's scale - format. group is the block of the operand, in its stored layout - (A [M, K], B [N, K]), that shares one scale; -1 spans the dimension: - [-1, -1] per-tensor, [1, -1] per-token, [-1, 1] per-channel, - [1, g] per-row groups of g along K, [r, c] 2-D blocks. - static=False means computed at runtime (activation quantization inside the op). - scale2: optional second-level scale (e.g. NVFP4's per-tensor fp32 global scale). - symmetric: False when the format carries zero points. - input: dtype the operand arrives in (default bf16). When it differs from - dtype, quantizing to dtype is part of the op; when equal, the operand - is pre-quantized and the op starts at the matmul. - An unquantized operand is {"dtype": "bf16"}. + scale: {"dtype", "static", "group": [rows, cols]}; group is the block of the stored + operand (A [M, K], B [N, K]) sharing one scale, -1 spans the dim: [-1, -1] per-tensor, + [1, -1] per-token, [-1, 1] per-channel, [1, g] groups of g along K, [r, c] 2-D blocks. + static=False: computed at runtime (quantization inside the op). + scale2: second-level scale (e.g. NVFP4's per-tensor fp32 global scale). + symmetric: False when the format has zero points. + input: dtype the operand arrives in (default bf16); != dtype means quantizing is in the op. + parallel: {"tp": T} - B split over K, partial C summed across T devices (in the op); + m, n, k are the full layer shape. """ m: int n: int @@ -103,6 +101,7 @@ def __post_init__(self): type="gemm", arg_schema=GemmArgs, description="C = activation(A[M,K] @ B[N,K]^T + bias)", + parallel_axes=parallel.GEMM_AXES, ) register(GEMM) diff --git a/operatorx/ops/moe.py b/operatorx/ops/moe.py index ad5f3c50b3..3644a6fccd 100644 --- a/operatorx/ops/moe.py +++ b/operatorx/ops/moe.py @@ -3,6 +3,7 @@ from dataclasses import dataclass from typing import Any +from operatorx.core import parallel from operatorx.core.op import OpSpec from operatorx.core.op_registry import register from operatorx.ops.gemm import ELEMENT_DTYPES, check_operand @@ -11,8 +12,7 @@ ACTIVATIONS = {"silu", "gelu", "gelu_tanh", "swigluoai", "situ", "swiglustep"} GATE_DTYPES = {"bf16", "fp32"} DISTRIBUTIONS = {"natural", "balanced", "single_hot"} -# vLLM's FusedMoEQuantConfig slots: a1 the experts' input, w1 the fused gate/up weight, -# w2 the down weight, a2 the intermediate activation +# vLLM's FusedMoEQuantConfig slots EXPERT_OPERANDS = ("a1", "w1", "w2", "a2") SHARED_OPERANDS = ("a1", "w1", "w2") @@ -125,33 +125,29 @@ def _check_routing(r: Any) -> None: class MoeArgs: """y[T,H] = shared(x) + sum over the selected experts e of w_e * expert_e(x). - The op starts at the router GEMM on normed hidden states x [T, H] and ends at - the combined output; residual add and norms are outside it. Operand - descriptors are the gemm op's ({"dtype", "scale"?, "scale2"?, "symmetric"?}). + From the router GEMM on normed x [T, H] to the combined output (residual/norms outside). + Operands are gemm operand descriptors. experts: {"num": E, "top_k": K, "inter": I, "a1", "w1", "w2", "a2", "bias"?: bool, "latent"?: L, "latent_norm"?: bool, "zero"?: n} - a1 is the experts' input as they consume it, w1 the fused gate/up weight - [2I, H], w2 the down weight [H, I], a2 the intermediate activation. latent: experts - run at width L with bf16 H->L / L->H projections (latent_norm: RMSNorm on the - routed latent output before L->H). zero: identity experts. + a1 expert input, w1 fused gate/up [2I, H], w2 down [H, I], a2 intermediate activation. + latent: experts at width L with bf16 H->L / L->H projections (latent_norm: RMSNorm + before L->H). zero: identity experts. router: {"gate": {"dtype", "logits"?}, "scoring": softmax|sigmoid|sqrtsoftplus, "select": {"kind": "topk"} | {"kind": "grouped_topk", "groups", "topk_groups"} | {"kind": "hash", "vocab"}, "bias"?: score-correction bias, "renormalize"?, "scale"?: routed scaling factor, "weight_on_input"?: router weight applied to the expert input} - gate.dtype is the router weight's dtype, gate.logits the dtype of the logits it - produces (default: gate.dtype). + gate.dtype: router weight dtype; gate.logits: logits dtype (default gate.dtype). activation: {"kind", "gated"?: default true, "interleaved"?: gate/up rows interleaved in w1 (default false: [gate; up]), "limit"?, "alpha"?, "beta"?} swigluoai: alpha scales the gate sigmoid, beta offsets up, limit clamps. situ: alpha and beta soft-cap the gate and up halves (alpha*tanh(g/alpha)*sigmoid(g) * beta*tanh(u/beta)). shared: null | {"count", "inter", "a1", "w1", "w2", "gate"?: null|"sigmoid"} routing: {"distribution": "natural" | "balanced" | "single_hot" | {"kind": "zipf", "s"}, "seed"} - natural: routing is whatever the router computes on seeded random inputs. - Otherwise expert choice is forced: balanced spreads tokens evenly over experts, - zipf draws experts with probability ~ 1/rank^s, single_hot sends every token to - expert 0 plus top_k-1 random experts. Router weights stay the router's. + natural: the router's own choice on seeded random inputs; otherwise forced: balanced + evenly, zipf p ~ 1/rank^s, single_hot expert 0 + top_k-1 random. Weights stay the router's. + parallel: {"tp", "dp", "ep"} - tokens is each dp group's batch; hidden/experts the full layer. """ tokens: int hidden: int @@ -179,6 +175,7 @@ def __post_init__(self): type="moe", arg_schema=MoeArgs, description="y = shared(x) + sum_{e in topk(route(x))} w_e * expert_e(x)", + parallel_axes=parallel.MOE_AXES, ) register(MOE) diff --git a/operatorx/recipes.py b/operatorx/recipes.py index b1058f52b3..82f98a378b 100644 --- a/operatorx/recipes.py +++ b/operatorx/recipes.py @@ -3,8 +3,9 @@ InferenceX serves every single-node configuration from a native srt-slurm recipe (inferencex-e2e/benchmarks/single_node/srt-slurm-recipes///-*/*.yaml). A case takes the recipe InferenceX runs for its source checkpoint on the runner's -hardware: the recipe's image, launch env and server arguments. InferenceX's own code -expands a recipe into its variants (infx.srt_slurm.synthetic_acceptance.selected_recipes). +hardware - the variant with the case's parallel split, else any of the checkpoint's - for +the recipe's image, launch env and server arguments. InferenceX's own code expands a +recipe into its variants (infx.srt_slurm.synthetic_acceptance.selected_recipes). Without a recipe the case runs on the backend's default image. """ from __future__ import annotations @@ -15,6 +16,8 @@ from pathlib import Path from typing import Any +from operatorx.core import parallel + REPO = Path(__file__).resolve().parents[1] / "inferencex-e2e" # infx, its recipes and srt-slurm RECIPES = REPO / "benchmarks/single_node/srt-slurm-recipes" FRAMEWORKS = ("vllm",) @@ -50,16 +53,33 @@ def _variants(framework: str, hardware: str) -> tuple[tuple[str, dict], ...]: return tuple(out) -def find(framework: str, hardware: str, checkpoints: list[str]) -> dict | None: - """The recipe InferenceX runs for the first of the checkpoints served on this hardware.""" +def _vllm_split(args: dict[str, Any]) -> dict: + tp, dp = int(args.get("tensor-parallel-size", 1)), int(args.get("data-parallel-size", 1)) + return {"tp": tp, "dp": dp, "ep": tp * dp if args.get("enable-expert-parallel") else 1, + "dcp": int(args.get("decode-context-parallel-size", 1))} + + +# per framework: the parallel split its serve arguments spell +SPLITS = {"vllm": _vllm_split} + + +def find(framework: str, hardware: str, checkpoints: list[str], split: dict | None = None, + axes: tuple[str, ...] = parallel.AXES) -> dict | None: + """The recipe InferenceX runs for the first of the checkpoints served on this hardware, + preferring a variant whose split equals the case's on `axes`.""" if framework not in FRAMEWORKS or not RECIPES.is_dir(): return None + want = parallel.normalize(split) variants = _variants(framework, hardware) for ckpt in checkpoints: - for key, r in variants: - if r["model"]["path"] == f"hf:{ckpt}": + mine = [(k, r) for k, r in variants if r["model"]["path"] == f"hf:{ckpt}"] + exact = [(k, r) for k, r in mine + if all(SPLITS[framework](r["roles"]["agg"]["args"])[a] == want[a] for a in axes)] + for match, pool in (("split", exact), ("checkpoint", mine)): + if pool: + key, r = pool[0] role = r["roles"]["agg"] - return {"recipe": key, "checkpoint": ckpt, "image": r["model"]["container"], + return {"recipe": key, "match": match, "checkpoint": ckpt, "image": r["model"]["container"], "env": {k: str(v) for k, v in (role.get("env") or {}).items()}, "engine_args": role.get("args") or {}} return None diff --git a/operatorx/runners/amd/backends/torch.py b/operatorx/runners/amd/backends/torch.py index 4f7bf08d9d..520f4e0821 100644 --- a/operatorx/runners/amd/backends/torch.py +++ b/operatorx/runners/amd/backends/torch.py @@ -1,4 +1,4 @@ -"""Dense GEMM through ROCm PyTorch, including architecture-correct FP8.""" +"""Dense GEMM through ROCm PyTorch (FP8 dtype chosen per gfx arch).""" from __future__ import annotations diff --git a/operatorx/runners/amd/backends/vllm.py b/operatorx/runners/amd/backends/vllm.py index 78cdf6235c..80dab41e40 100644 --- a/operatorx/runners/amd/backends/vllm.py +++ b/operatorx/runners/amd/backends/vllm.py @@ -1,7 +1,4 @@ -"""Dense GEMM, MoE and attention modules through vLLM's own layers and kernel selection (ROCm). - -AITER is enabled as in InferenceX's ROCm vLLM launches. -""" +"""GEMM, MoE and attention through vLLM's layers (ROCm), AITER on as in InferenceX's ROCm launches.""" import os os.environ.setdefault("VLLM_ROCM_USE_AITER", "1") diff --git a/operatorx/runners/amd/runner.py b/operatorx/runners/amd/runner.py index 7b5d66aaae..7384019aaf 100644 --- a/operatorx/runners/amd/runner.py +++ b/operatorx/runners/amd/runner.py @@ -9,57 +9,16 @@ import torch from operatorx.core import Op, Result, UnsupportedOpError -from operatorx.runners.common import profiling, telemetry +from operatorx.runners.common import profiling, ranks, telemetry, timing -_WARMUP = 5 -_ITERS = 10 -# Same timing protocol as the NVIDIA runner: wall-clock warmup floor, GPU -# spin ahead of the timed loop, per-iteration cache flush, inter-op cooldown. -_WARMUP_MIN_S = float(os.environ.get("OPERATORX_WARMUP_MIN_S", "0.025")) -# The first op of a process starts from an idle GPU; give it a longer ramp. -_FIRST_WARMUP_S = float(os.environ.get("OPERATORX_FIRST_WARMUP_S", "1.0")) -_FIRST = True -_SHIELD_CYCLES = int(os.environ.get("OPERATORX_SHIELD_CYCLES", "4000000")) _COOLDOWN_RATIO = float(os.environ.get("OPERATORX_COOLDOWN_RATIO", "4")) _COOLDOWN_MAX_S = float(os.environ.get("OPERATORX_COOLDOWN_MAX_S", "1.0")) -# ROCm reports only the per-XCD L2 slice, so the flush buffer is sized to -# cover the whole cache hierarchy. +# ROCm reports only the per-XCD L2 slice: size the flush buffer for the whole hierarchy _FLUSH_MB = int(os.environ.get("OPERATORX_FLUSH_MB", "512")) _L2_BUF: dict[int, torch.Tensor] = {} -def _time_op(fn, device: int, sleep_s: float) -> float: - """Median of _ITERS cold, event-timed iterations in us.""" - for _ in range(_WARMUP): - fn() - torch.cuda.synchronize() - global _FIRST - floor, _FIRST = (max(_WARMUP_MIN_S, _FIRST_WARMUP_S) if _FIRST else _WARMUP_MIN_S), False - t0 = time.perf_counter() - while time.perf_counter() - t0 < floor: - fn() - torch.cuda.synchronize() - - starts = [torch.cuda.Event(enable_timing=True) for _ in range(_ITERS)] - ends = [torch.cuda.Event(enable_timing=True) for _ in range(_ITERS)] - if sleep_s <= 0.0 and _SHIELD_CYCLES > 0: - torch.cuda._sleep(_SHIELD_CYCLES) - for start, end in zip(starts, ends): - if sleep_s > 0.0: - time.sleep(sleep_s) - torch.cuda._sleep(max(_SHIELD_CYCLES // 4, 500000)) - _L2_BUF[device].zero_() - start.record() - fn() - end.record() - if sleep_s > 0.0: - torch.cuda.synchronize() - torch.cuda.synchronize() - times = sorted(s.elapsed_time(e) * 1000.0 for s, e in zip(starts, ends)) - return times[_ITERS // 2] - - def run(op: Op) -> Result: if op.backend not in ("torch", "vllm"): raise UnsupportedOpError(f"unknown AMD backend: {op.backend}") @@ -80,10 +39,10 @@ def run(op: Op) -> Result: fn, cuda_graph = impl.launcher(ctx) if impl.launcher else ((lambda: impl.kernel(ctx)), False) median_us, telem = telemetry.measure( - op, lambda sleep_s: _time_op(fn, device, sleep_s)) + op, lambda sleep_s: timing.time_op(fn, _L2_BUF[device].zero_, sleep_s)) if _COOLDOWN_RATIO > 0.0: - time.sleep(min(median_us * 1e-6 * (_ITERS + _WARMUP) * _COOLDOWN_RATIO, + time.sleep(min(median_us * 1e-6 * (timing.ITERS + timing.WARMUP) * _COOLDOWN_RATIO, _COOLDOWN_MAX_S)) metrics = {"latency_us": median_us, "cuda_graph": cuda_graph, "telemetry": telem} if isinstance(ctx, dict) and ctx.get("meta"): @@ -91,4 +50,9 @@ def run(op: Op) -> Result: prof = profiling.profile_op(fn) if prof is not None: metrics["profile"] = prof + per_rank = ranks.summary(metrics) # every rank's timeline and telemetry, on rank 0 + if per_rank is not None: + metrics["ranks"] = per_rank + if per_rank["capped_ranks"] and metrics.get("telemetry"): + metrics["telemetry"]["capped"] = True # any rank capped caps the op return Result(op=op, metrics=metrics) diff --git a/operatorx/runners/common/collectivex.py b/operatorx/runners/common/collectivex.py new file mode 100644 index 0000000000..bf699d125d --- /dev/null +++ b/operatorx/runners/common/collectivex.py @@ -0,0 +1,23 @@ +"""CollectiveX modules OperatorX reuses as-is (the sibling collectivex/ tree).""" +from __future__ import annotations + +import functools +import importlib.util +import sys +from pathlib import Path + +_ROOT = Path(__file__).resolve().parents[3] / "collectivex" + + +def _load(name: str, rel: str): + spec = importlib.util.spec_from_file_location(name, _ROOT / rel) + module = importlib.util.module_from_spec(spec) + sys.modules[name] = module # its dataclasses resolve their module by name + spec.loader.exec_module(module) + return module + + +@functools.cache +def harness(): + return _load("collectivex_ep_harness", "bench/ep_harness.py") + diff --git a/operatorx/runners/common/profiling.py b/operatorx/runners/common/profiling.py index 8511c2ba09..d22142ca1a 100644 --- a/operatorx/runners/common/profiling.py +++ b/operatorx/runners/common/profiling.py @@ -1,34 +1,11 @@ -"""Per-op kernel decomposition, run after timing so it cannot affect latency_us. - -Each op is replayed under torch.profiler - kineto with CUPTI on CUDA, -rocprofiler on ROCm - and the per-kernel breakdown is attached to the result: - - metrics["profile"] = { - "iters": N, "op_index": i, - "kernels": [{"name", "cat", "count_per_call", "us_per_call", - "grid", "block", "regs", "smem", "blocks_per_sm", - "warps_per_sm", "occupancy_pct"}, ...], # sorted by time - "gpu_us_per_call": ..., # sum of kernel durations - "span_us", "busy_us", "gap_us", "overlap_us", "streams", - # of the median-span replay: first start -> last end, - # union of kernel time across streams, span - busy, - # sum of durations - busy (concurrent kernels) - "timeline": [{"name", "stream", "start_us", "dur_us"}, ...], # that replay, op-relative - "flush_kernels_excluded": ..., - "trace": ..., # when a chrome trace was kept - } - -Launch-config fields are present only where the platform reports them. - -OPERATORX_PROFILE=0 disable -OPERATORX_PROFILE_ITERS replays per op (default 3) -OPERATORX_PROFILE_FLUSH_MB flush size before each replay (default 512, 0 = warm) -OPERATORX_PROFILE_TRACE_DIR keep chrome traces here -OPERATORX_PROFILE_TRACE_EVERY keep every Nth op's trace (default 200) -OPERATORX_PROFILE_MARKERS=1 instead of torch.profiler, wrap the replays in an - nvtx/roctx range "opx" for an external - profiler (ncu, rocprofv3); op_index is the join key. - Latencies from such runs are not timing data. +"""Per-op kernel breakdown under torch.profiler (CUPTI / rocprofiler), after timing. + +metrics["profile"]: iters, op_index, kernels [{name, cat, count_per_call, us_per_call, grid, block, +regs, smem, blocks_per_sm, warps_per_sm, occupancy_pct}] (by time; launch fields where reported), +gpu_us_per_call, span/busy/gap/overlap_us, streams, timeline (median-span replay), trace. + +Env OPERATORX_PROFILE_*: =0 off; _ITERS (3); _FLUSH_MB per replay (512, 0 = warm); _TRACE_DIR, +_TRACE_EVERY (200); _MARKERS=1: nvtx/roctx range "opx" for ncu/rocprofv3 instead. """ from __future__ import annotations @@ -48,8 +25,8 @@ _MARKERS = os.environ.get("OPERATORX_PROFILE_MARKERS", "") == "1" # name of the kernel the int8 zero_() flush dispatches _FLUSH_KERNEL_MARKER = "FillFunctor" -# GPU spin enqueued after each flush so the replay's launches queue up behind it and -# the timeline shows device time, not host launch gaps; torch.cuda._sleep's kernel +# spin after each flush so launches queue behind it and the timeline shows device time, +# not host gaps; torch.cuda._sleep's kernel _SHIELD_CYCLES = int(os.environ.get("OPERATORX_SHIELD_CYCLES", "4000000")) _SHIELD_KERNEL_MARKER = "spin_kernel" @@ -102,9 +79,8 @@ def name_of(self, e): return e.get("name", "")[:200] def harness_events(self, events: list[dict]) -> tuple[set[int], set[int]]: - """(flush, spin) event ids among the sorted device events. The op itself may launch - fill kernels, so the flush is identified by position: the last fill before each spin - (the loop issues flush, spin, replay). Without a spin, every fill counts as a flush.""" + """(flush, spin) event ids. The op may launch fills itself, so the flush is the last fill + before each spin (flush, spin, replay); without a spin every fill is a flush.""" flush, spin, last_fill = set(), set(), None for i, e in enumerate(events): name = e.get("name", "") @@ -142,13 +118,12 @@ def decompose(self, events: list[dict], iters: int) -> tuple[list[dict], list[li def _platform(): - """The capture for the device in front of us: CUDA and ROCm share one.""" return _Cuda() def _markers_pass(kernel_fn, plat) -> dict: marker = f"opx{_counter:06d}" - plat.flush() # outside the range so the flush is not attributed + plat.flush() # outside the range: not attributed torch.cuda.synchronize() torch.cuda.nvtx.range_push(marker) try: diff --git a/operatorx/runners/common/ranks.py b/operatorx/runners/common/ranks.py new file mode 100644 index 0000000000..ff8fd06fbf --- /dev/null +++ b/operatorx/runners/common/ranks.py @@ -0,0 +1,100 @@ +"""Agreement and reduction across ranks; no-ops on one rank. + +Every rank must take the same branch wherever a collective launches, or the others hang. +""" +from __future__ import annotations + +from contextlib import contextmanager + +import torch + +from operatorx.core import UnsupportedOpError + + +def _group(): + import torch.distributed as dist + return dist if dist.is_available() and dist.is_initialized() and dist.get_world_size() > 1 else None + + +def world() -> int: + d = _group() + return d.get_world_size() if d else 1 + + +def agree(ok: bool) -> bool: + """True only when every rank passes True.""" + d = _group() + if d is None: + return ok + t = torch.tensor([1 if ok else 0], dtype=torch.int32, device="cuda") + d.all_reduce(t, op=d.ReduceOp.MIN) + return bool(t.item()) + + +def any_(flag: bool) -> bool: + return not agree(not flag) + + +_TOKEN: dict[int, torch.Tensor] = {} + + +def align(cycles: int) -> None: + """Device-side rank barrier then a spin, without a host sync (CollectiveX _graph_align).""" + d = _group() + if d is None: + return + dev = torch.cuda.current_device() + if dev not in _TOKEN: + _TOKEN[dev] = torch.zeros(1, device="cuda") + d.all_reduce(_TOKEN[dev]) + torch.cuda._sleep(cycles) + + +@contextmanager +def together(what: str): + """Every rank reports once whether it got through; all stop if any did not.""" + try: + yield + except BaseException: + agree(False) + raise + if not agree(True): + raise UnsupportedOpError(f"another rank could not {what}") + + +LAST: dict = {} # the last timing's reduction, for summary() + + +def iterations(times: list[float]) -> dict: + """Per-iteration latencies reduced across ranks (CollectiveX ep_harness._reduce_vec).""" + d = _group() + if d is None: + out = {"max": list(times), "min": list(times), "spread": [0.0] * len(times)} + else: + from operatorx.runners.common import collectivex + h = collectivex.harness() + dev = torch.device("cuda", torch.cuda.current_device()) + hi = h._reduce_vec(torch, d, dev, times, d.ReduceOp.MAX) + lo = h._reduce_vec(torch, d, dev, times, d.ReduceOp.MIN) + out = {"max": hi, "min": lo, "spread": [a - b for a, b in zip(hi, lo)]} + LAST.clear() + LAST.update(out, own=list(times)) + return out + + +def summary(metrics: dict) -> dict | None: + """Every rank's latency, timeline and telemetry, gathered on rank 0; None on one device.""" + d = _group() + if d is None: + return None + + def med(xs): + return sorted(xs)[len(xs) // 2] if xs else None + + mine = {"rank": d.get_rank(), "latency_us": med(LAST.get("own", [])), + "profile": metrics.get("profile"), "telemetry": metrics.get("telemetry")} + gathered = [None] * d.get_world_size() if d.get_rank() == 0 else None + d.gather_object(mine, gathered, dst=0) + capped = [p["rank"] for p in gathered or () if (p.get("telemetry") or {}).get("capped")] + return {"world": d.get_world_size(), "latency_us_min": med(LAST.get("min", [])), + "skew_us": med(LAST.get("spread", [])), "capped_ranks": capped, "per_rank": gathered} diff --git a/operatorx/runners/common/telemetry.py b/operatorx/runners/common/telemetry.py index 13b8410e1c..70f8abfad5 100644 --- a/operatorx/runners/common/telemetry.py +++ b/operatorx/runners/common/telemetry.py @@ -1,25 +1,13 @@ -"""Per-op GPU telemetry and the power-throttle retry policy. - -A provider polls clocks, power, temperatures and throttle reasons every -OPERATORX_TELEMETRY_MS on a background thread while an op is timed, and -reads cumulative counters (energy, time spent in each throttle reason) at -the start and end of the window. An attempt is capped when a power/thermal -throttle reason is active while the SM clock sits more than -OPERATORX_CAP_CLOCK_FRACTION below rated boost; the power-cap reason at -full boost is ordinary DVFS and does not count. Capped attempts are -retried with growing inter-kernel sleeps (OPERATORX_RETRY_SLEEP_MS, -doubling) up to OPERATORX_THROTTLE_RETRIES times, and the attempt with -the lowest median is kept. - -metrics["telemetry"] = { - "provider", "interval_ms", "n_samples", "window_ms", "rated_sm_clock_mhz", - : {"min", "p50", "max"}, ... # sm_clock_mhz, power_w, ... - "counters": {...}, # deltas over the window (100 ms steps) - "throttle_reasons", "capped", "attempts", "inter_kernel_sleep_ms", -} +"""Per-op GPU telemetry (sampled every OPERATORX_TELEMETRY_MS) and the throttle retry. + +An attempt is capped when a power/thermal reason is active with the SM clock below +OPERATORX_CAP_CLOCK_FRACTION x rated boost (power cap at full boost is ordinary DVFS); +capped attempts retry with doubling sleeps (OPERATORX_RETRY_SLEEP_MS, up to +OPERATORX_THROTTLE_RETRIES) and the lowest median is kept. -With OPERATORX_TELEMETRY_DIR set, the raw samples of every attempt are -appended to /telemetry-rank.jsonl. +metrics["telemetry"]: provider, interval_ms, n_samples, window_ms, rated_sm_clock_mhz, +: {min, p50, max}, counters (window deltas), throttle_reasons, capped, attempts, +inter_kernel_sleep_ms. Raw samples: OPERATORX_TELEMETRY_DIR/telemetry-rank.jsonl. """ from __future__ import annotations @@ -30,6 +18,8 @@ import torch +from operatorx.runners.common import ranks + _INTERVAL_MS = float(os.environ.get("OPERATORX_TELEMETRY_MS", "5")) _DIR = os.environ.get("OPERATORX_TELEMETRY_DIR") _CAP_CLOCK_FRACTION = float(os.environ.get("OPERATORX_CAP_CLOCK_FRACTION", "0.985")) @@ -64,13 +54,11 @@ "below_app_clocks_ms": "NVML_FI_DEV_PERF_POLICY_TOTAL_APP_CLOCKS", "below_base_clocks_ms": "NVML_FI_DEV_PERF_POLICY_TOTAL_BASE_CLOCKS", } -# The driver advances these counters (and energy) in 100 ms steps, so deltas -# are only meaningful for windows much longer than that. +# the driver advances these counters (and energy) in 100 ms steps _COUNTER_MIN_WINDOW_NS = 1_000_000_000 class _Provider: - """Samples are dicts with "t_ns" plus whichever fields the device reports.""" name = "null" rated_mhz: int | None = None @@ -131,7 +119,7 @@ def __init__(self, device: int) -> None: super().__init__() import pynvml as nv nv.nvmlInit() - # NVML ignores CUDA_VISIBLE_DEVICES, so match the device by UUID. + # NVML ignores CUDA_VISIBLE_DEVICES: match by UUID uuid = str(torch.cuda.get_device_properties(device).uuid) self._nv = nv self._h = nv.nvmlDeviceGetHandleByUUID(uuid if uuid.startswith("GPU-") else f"GPU-{uuid}") @@ -207,8 +195,7 @@ def __init__(self, device: int) -> None: super().__init__() import amdsmi amdsmi.amdsmi_init() - # amdsmi enumerates every GPU regardless of *_VISIBLE_DEVICES, so - # match the device by PCI address. + # amdsmi ignores *_VISIBLE_DEVICES: match by PCI address p = torch.cuda.get_device_properties(device) bdf = f"{p.pci_domain_id:04x}:{p.pci_bus_id:02x}:{p.pci_device_id:02x}." self._a = amdsmi @@ -305,9 +292,7 @@ def _dump(op, attempt: int, summary: dict, samples: list[dict]) -> None: def measure(op, time_once) -> tuple[float, dict]: - """Time op with time_once(sleep_s) -> median_us, retrying capped attempts. - - Returns the best median and its telemetry summary.""" + """time_once(sleep_s) -> median_us, retried while capped; the best median and its telemetry.""" provider = _provider() best = None for attempt in range(1 + max(_RETRIES, 0)): @@ -318,8 +303,7 @@ def measure(op, time_once) -> tuple[float, dict]: try: median_us = time_once(sleep_s) finally: - # a kernel that faults mid-replay must not leave the sampler polling into - # the next op's telemetry + # a faulting kernel must not leave the sampler polling into the next op samples = provider.stop() t1 = time.time_ns() after = provider.counters() @@ -328,7 +312,7 @@ def measure(op, time_once) -> tuple[float, dict]: _dump(op, attempt, summary, samples) if best is None or median_us < best[0]: best = (median_us, summary, sleep_s) - if not summary["capped"]: + if not ranks.any_(summary["capped"]): # every rank retries, or none does break median_us, summary, sleep_s = best summary["attempts"] = attempt + 1 diff --git a/operatorx/runners/common/timing.py b/operatorx/runners/common/timing.py new file mode 100644 index 0000000000..76f1789b56 --- /dev/null +++ b/operatorx/runners/common/timing.py @@ -0,0 +1,56 @@ +"""Event timing of one op on one or many devices. + +Per iteration: flush, rank align, start, op, end. A spin ahead of the loop lets the host +enqueue every iteration first; multi-device iterations start behind a device-side barrier ++ spin (CollectiveX _graph_align) so the cross-rank max is the op, not launch skew. +""" +from __future__ import annotations + +import os +import time + +import torch + +from operatorx.runners.common import ranks + +WARMUP = 5 +ITERS = 10 +_WARMUP_MIN_S = float(os.environ.get("OPERATORX_WARMUP_MIN_S", "0.025")) +_FIRST_WARMUP_S = float(os.environ.get("OPERATORX_FIRST_WARMUP_S", "1.0")) +_SHIELD_CYCLES = int(os.environ.get("OPERATORX_SHIELD_CYCLES", "100000000")) # ~50 ms +_ALIGN_CYCLES = int(os.environ.get("OPERATORX_ALIGN_CYCLES", "10000000")) # ~5 ms +_RETRY_SPIN_CYCLES = 2_000_000 +_FIRST = True + + +def time_op(fn, flush, sleep_s: float) -> float: + """Median over ITERS cold iterations of fn, in us (each iteration's slowest rank).""" + global _FIRST + for _ in range(WARMUP): + fn() + torch.cuda.synchronize() + floor, _FIRST = (max(_WARMUP_MIN_S, _FIRST_WARMUP_S) if _FIRST else _WARMUP_MIN_S), False + t0 = time.perf_counter() + while ranks.any_(time.perf_counter() - t0 < floor): # same round count on every rank + fn() + torch.cuda.synchronize() + + align = _ALIGN_CYCLES if ranks.world() > 1 else 0 + starts = [torch.cuda.Event(enable_timing=True) for _ in range(ITERS)] + ends = [torch.cuda.Event(enable_timing=True) for _ in range(ITERS)] + if sleep_s <= 0.0: + torch.cuda._sleep(_SHIELD_CYCLES) + for start, end in zip(starts, ends): + if sleep_s > 0.0: # throttle retry + time.sleep(sleep_s) + torch.cuda._sleep(_RETRY_SPIN_CYCLES) + flush() + ranks.align(align) + start.record() + fn() + end.record() + if sleep_s > 0.0: + torch.cuda.synchronize() + torch.cuda.synchronize() + times = [s.elapsed_time(e) * 1000.0 for s, e in zip(starts, ends)] + return sorted(ranks.iterations(times)["max"])[ITERS // 2] diff --git a/operatorx/runners/common/trace.py b/operatorx/runners/common/trace.py index 5e31ecdb55..ee7e35cda6 100644 --- a/operatorx/runners/common/trace.py +++ b/operatorx/runners/common/trace.py @@ -1,11 +1,5 @@ -"""Chrome-trace arithmetic shared by the per-platform profilers. - -Platforms differ in how they capture (CUPTI, rocprofiler) -and in how a replay is delimited, but once a replay is a list of device events -with a start, a duration and a lane, the structure of a measurement - which -kernels ran, how long the device was busy, where it was idle, what overlapped - -is the same arithmetic everywhere. -""" +"""Chrome-trace arithmetic shared by the per-platform profilers (CUPTI, rocprofiler): +a replay is a list of device events with start, duration and lane.""" from __future__ import annotations import json @@ -39,12 +33,8 @@ def busy_us(events: list[dict]) -> float: def replay_stats(replays: list[list[dict]], stream_of: Callable[[dict], object]) -> dict | None: - """Timing structure of the replays, reported from the median one by span. - - span is first start to last end, busy the union of device time across lanes, - gap the device idle inside the span, and overlap the time counted twice - because lanes ran concurrently. - """ + """Median replay by span: span first start to last end, busy the union of device time, + gap idle inside the span, overlap time counted twice by concurrent lanes.""" rows = [] for r in replays: if not r: diff --git a/operatorx/runners/common/vllm/__init__.py b/operatorx/runners/common/vllm/__init__.py index 802e7999ed..01d16d5385 100644 --- a/operatorx/runners/common/vllm/__init__.py +++ b/operatorx/runners/common/vllm/__init__.py @@ -1 +1 @@ -"""vLLM layers as operatorx backends: dense GEMM (linear) and MoE layers (moe).""" +"""vLLM layers as operatorx backends: linear (GEMM), moe, attention.""" diff --git a/operatorx/runners/common/vllm/attention.py b/operatorx/runners/common/vllm/attention.py index cc0d01f692..700b73e7e4 100644 --- a/operatorx/runners/common/vllm/attention.py +++ b/operatorx/runners/common/vllm/attention.py @@ -1,12 +1,8 @@ """Attention modules through vLLM's own model code, KV cache and scheduling. -An op picks the checkpoint family whose module it describes (attention_models.json holds -each family's config.json), overrides the module's sizes, cuts the model to the layers -that module needs, and shrinks the MLP. vLLM builds that model with dummy weights, -allocates and lays out its KV cache, and picks the attention backend. The op's batch is -scheduled through vLLM's model runner as requests whose ctx tokens are already computed -(the cache holds random, format-valid data); the timed call is the module's forward, -with the arguments and forward context the model's own decoder layer gives it. +The op's checkpoint family (attention_models.json) with its sizes, cut to the layers the module +needs, dummy weights; the batch is scheduled through the model runner with ctx tokens already +computed, and the module's forward is timed with its decoder layer's arguments and forward context. """ from __future__ import annotations @@ -23,6 +19,8 @@ import torch from operatorx.core import BackendImpl, Op, UnsupportedOpError +from operatorx.runners.common import ranks +from operatorx.runners.common.vllm import engine as vllm_engine from operatorx.runners.common.vllm import linear as vllm_linear _MODELS = json.loads((Path(__file__).with_name("attention_models.json")).read_text()) @@ -60,8 +58,7 @@ def _yarn(s: dict | None) -> dict | None: def _quant(op: Op, family: str) -> dict | None: - """The quantization_config of the family's checkpoint whose projections carry the op's - operands (proj lists the quantized projections; the rest stay bf16).""" + """The family checkpoint variant's quantization_config whose projections match proj.""" proj = op.args.get("proj") or {} if not proj: return None @@ -104,8 +101,7 @@ def _build_mla(op: Op) -> _Build: def _qwen35(a: dict, full: dict, linear: dict, target: str) -> _Build: - """Qwen3.5: one Gated DeltaNet layer then one gated full-attention layer, so the KV - cache has the hybrid layout serving uses.""" + """Qwen3.5: one Gated DeltaNet then one gated full-attention layer (serving's hybrid KV layout).""" c = _family("qwen3_5_moe") t = c["text_config"] t.update(num_hidden_layers=2, layer_types=["linear_attention", "full_attention"], mtp_num_hidden_layers=0, @@ -200,9 +196,8 @@ def _build_qsa(op: Op) -> _Build: indexer_compress_ratio=a["compress"]), "layers.1.self_attn") -# DeepSeek-V4.1's layer roles, as the checkpoint arranges them: window-only, then ratio-2 -# (a KV + index source, a consumer reusing its top-k), then ratio-1 (the KV + index source -# that publishes candidate blocks, a consumer, an index source over the shared index cache). +# DeepSeek-V4.1's layer roles, in checkpoint order: window-only; ratio-2 source, top-k reuser; +# ratio-1 source (publishes candidate blocks), consumer, index source over the shared index cache _V41_RATIOS = [0, 2, 2, 1, 1, 1] _V41_LAYER = {(0, True, None): 0, (2, True, "own"): 1, (2, False, "reuse"): 2, (1, True, "own"): 3, (1, False, "reuse"): 4, (1, False, "shared"): 5} @@ -276,28 +271,26 @@ def _build_gdn_qwen38(a: dict, linear: dict) -> _Build: class _Engine: """One vLLM engine (in-process) for one module config; reused while ops share it.""" - def __init__(self, key: str, b: _Build): + def __init__(self, key: str, b: _Build, split: dict | None): os.environ["VLLM_ENABLE_V1_MULTIPROCESSING"] = "0" from vllm import LLM self.key = key self.dir = tempfile.mkdtemp(prefix="opx-attn-") Path(self.dir, "config.json").write_text(json.dumps(b.config)) - recipe, self.recipe = _recipe_kwargs(b.without) - kwargs = {"max_num_batched_tokens": _MAX_BATCHED_TOKENS, **recipe} - kwargs.update(model=self.dir, load_format="dummy", skip_tokenizer_init=True, enforce_eager=True, - enable_prefix_caching=False, max_num_seqs=_MAX_SEQS, kv_cache_memory_bytes=_kv_bytes(), - gpu_memory_utilization=_GPU_UTIL) - # compile kernels when first used (the untimed step) rather than every shape up - # front; FlashInfer autotuning still runs + vllm_engine.launch() + kwargs, self.recipe = vllm_engine.engine_args( + split, without=b.without, eager=True, defaults={"max_num_batched_tokens": _MAX_BATCHED_TOKENS}, + model=self.dir, load_format="dummy", skip_tokenizer_init=True, enforce_eager=True, + enable_prefix_caching=False, max_num_seqs=_MAX_SEQS, kv_cache_memory_bytes=_kv_bytes(), + gpu_memory_utilization=_GPU_UTIL, **b.engine) + # JIT kernels on first use (the untimed step), not every shape up front kernel = kwargs.get("kernel_config") if kernel is None or isinstance(kernel, dict): kwargs["kernel_config"] = {**(kernel or {}), "enable_jit_warmup": False} else: kernel.enable_jit_warmup = False - attention = {**self.recipe["attention_config"], **b.engine.pop("attention_config", {})} - kwargs.update(b.engine) - if attention: - kwargs["attention_config"] = attention + # the engine runs eagerly: the full graphs serving would capture, from vLLM + self.capture = vllm_engine.full_graph_sizes(kwargs, self.recipe["compilation_config"]) self.reqs: list = [] self.n = 0 try: @@ -311,7 +304,8 @@ def __init__(self, key: str, b: _Build): raise def close(self) -> None: - from vllm.distributed.parallel_state import cleanup_dist_env_and_memory + # keep the process group for the next engine + from vllm.distributed.parallel_state import destroy_model_parallel try: self.release() if hasattr(self, "llm"): @@ -320,7 +314,7 @@ def close(self) -> None: print(f"[vllm.attention] engine shutdown: {type(e).__name__}: {e}", file=sys.stderr) for k in ("llm", "runner", "kvm"): self.__dict__.pop(k, None) - cleanup_dist_env_and_memory() + destroy_model_parallel() gc.collect() torch.cuda.empty_cache() @@ -334,7 +328,6 @@ def _output(self, new: list, scheduled: dict, finished: set): finished_req_ids=finished, free_encoder_mm_hashes=[]) def _configured(self): - """vLLM's current config, as the worker sets it around the model runner's step.""" from vllm.config import set_current_vllm_config return set_current_vllm_config(self.runner.vllm_config) @@ -349,8 +342,7 @@ def release(self) -> None: self.reqs = [] def step(self, batch: dict, path: str) -> dict: - """Schedule the batch; return the call the model made to the module at path (a - module-name suffix), with its forward context.""" + """Schedule the batch; return the model's call to the module at path, with its forward context.""" from vllm import SamplingParams from vllm.forward_context import get_forward_context from vllm.v1.core.sched.output import NewRequestData @@ -365,8 +357,7 @@ def step(self, batch: dict, path: str) -> dict: ctx = g["ctx"] if isinstance(g["ctx"], int) else rng.randint(g["ctx"]["min"], g["ctx"]["max"]) self.n += 1 r = Request(f"opx{self.n}", [0] * (ctx + g["q"]), SamplingParams(max_tokens=1), None) - # as the scheduler allocates a request whose ctx tokens are computed: every - # block for full attention, only the window's for sliding-window caches + # as the scheduler allocates computed ctx: every block (full), window only (sliding) r.num_computed_tokens = ctx self.reqs.append(r) if self.kvm.allocate_slots(r, g["q"]) is None: @@ -397,35 +388,6 @@ def capture(*args, **kwargs): return seen -def _recipe_kwargs(without: tuple = ()) -> tuple[dict, dict]: - """The InferenceX recipe's server arguments (OPERATORX_ENGINE_ARGS, set per shard by the - planner from operatorx.recipes) as vLLM EngineArgs, parsed by vLLM's own parser; and - what the engine itself does not apply: the attention config (merged with the op's) and - the CUDA-graph mode and sizes serving would use (the engine here runs eagerly).""" - import dataclasses - - from vllm.engine.arg_utils import EngineArgs - from vllm.utils.argparse_utils import FlexibleArgumentParser - - from operatorx.recipes import serve_argv - args = {k: v for k, v in json.loads(os.environ.get("OPERATORX_ENGINE_ARGS") or "{}").items() if k not in without} - attention = json.loads(args.pop("attention-config", None) or "{}") - compilation = json.loads(args.pop("compilation-config", None) or "{}") - if args.get("max-cudagraph-capture-size"): - compilation.setdefault("max_cudagraph_capture_size", int(args["max-cudagraph-capture-size"])) - info = {"attention_config": attention, "compilation_config": compilation} - if not args: - return {}, info - parser = EngineArgs.add_cli_args(FlexibleArgumentParser()) - base = vars(parser.parse_known_args(["--model", "m"])[0]) - ns = vars(parser.parse_known_args(["--model", "m", *serve_argv(args)])[0]) - fields = {f.name for f in dataclasses.fields(EngineArgs)} - {"model"} - kwargs = {k: v for k, v in ns.items() if k in fields and v != base.get(k)} - if compilation.get("custom_ops"): # custom-op selection holds without compilation - kwargs["compilation_config"] = {"custom_ops": compilation["custom_ops"]} - return kwargs, info - - def _kv_bytes() -> int: return int(torch.cuda.get_device_properties(torch.cuda.current_device()).total_memory * _KV_FRACTION) @@ -443,10 +405,8 @@ def _shuffle(blocks: list, rng: random.Random) -> list: def _fill_caches(runner) -> None: - """Random, format-valid contents for every KV cache and state buffer. Caches are - views (often several dtypes, packed layouts with scales) over raw byte storage; every - byte is drawn below 0x40, which decodes to a small finite value in fp32, bf16, fp8 - e4m3 / ue8m0 and int8 alike.""" + """Random bytes below 0x40 in every KV cache/state buffer: small and finite as fp32, bf16, + fp8 e4m3 / ue8m0 and int8 alike (caches are multi-dtype views over raw bytes).""" ctx = runner.vllm_config.compilation_config.static_forward_context seen: set[int] = set() with torch.no_grad(): @@ -467,15 +427,15 @@ def _fill_caches(runner) -> None: _ENGINE: _Engine | None = None -def _engine(b: _Build) -> _Engine: +def _engine(b: _Build, split: dict | None) -> _Engine: global _ENGINE - key = json.dumps([b.family, b.config, b.engine], sort_keys=True) + key = json.dumps([b.family, b.config, b.engine, split], sort_keys=True) if _ENGINE is not None and _ENGINE.key == key: return _ENGINE if _ENGINE is not None: _ENGINE.close() _ENGINE = None - _ENGINE = _Engine(key, b) + _ENGINE = _Engine(key, b, split) return _ENGINE @@ -495,21 +455,24 @@ def _prepare(op: Op) -> dict: kv = a.get("kv_cache_dtype") if kv is not None: b.engine["kv_cache_dtype"] = _KV_DTYPES[kv] - try: - eng = _engine(b) - # vLLM's own startup (model build, profiling and warmup runs) failing on this config - except (ValueError, NotImplementedError, AssertionError, RuntimeError) as e: - if vllm_linear._is_fault(e): - raise - raise UnsupportedOpError(f"vLLM rejected the {b.family} module: {type(e).__name__}: {e}"[:400]) from e - try: - seen = eng.step(a["batch"], b.module) - except (ValueError, NotImplementedError, AssertionError) as e: - if vllm_linear._is_fault(e): - raise - raise UnsupportedOpError(f"vLLM rejected the batch: {type(e).__name__}: {e}"[:400]) from e + split = a.get("parallel") + with ranks.together("build this engine"): + try: + eng = _engine(b, split) + # vLLM's startup (build, profiling, warmup) failing on this config + except (ValueError, NotImplementedError, AssertionError, RuntimeError) as e: + if vllm_linear._is_fault(e): + raise + raise UnsupportedOpError(f"vLLM rejected the {b.family} module: {type(e).__name__}: {e}"[:400]) from e + with ranks.together("schedule this batch"): + try: + seen = eng.step(a["batch"], b.module) + except (ValueError, NotImplementedError, AssertionError) as e: + if vllm_linear._is_fault(e): + raise + raise UnsupportedOpError(f"vLLM rejected the batch: {type(e).__name__}: {e}"[:400]) from e ctx = {"engine": eng, **seen, - "meta": {"vllm_family": b.family, "vllm_repo": _MODELS[b.family]["repo"], "vllm_module": b.module, + "meta": {"vllm_engine": vllm_engine.meta(), "vllm_family": b.family, "vllm_repo": _MODELS[b.family]["repo"], "vllm_module": b.module, "vllm_backends": _backends(eng.runner), "vllm_attn_metadata": {k: type(v).__name__ for k, v in (seen["fc"].attn_metadata or {}).items()}, "kv_cache_groups": [ @@ -522,8 +485,7 @@ def _prepare(op: Op) -> dict: def _replaying(ctx: dict): - """The context the model runner's step gives a forward: inference mode (its tensors - are inference tensors), vLLM's current config, and the step's forward context.""" + """The model runner step's context: inference mode, vLLM's current config, forward context.""" import contextlib from vllm.config import set_current_vllm_config @@ -541,8 +503,7 @@ def _kernel(ctx: dict) -> None: def _cudagraph(ctx: dict) -> bool: - """Whether vLLM would replay this batch as a full CUDA graph: a uniform decode batch - within the capture sizes, on backends that support one.""" + """vLLM would replay this as a full CUDA graph: uniform decode within capture sizes, backend support.""" with ctx["engine"]._configured(): # backends read vLLM's current config to answer return _full_graph(ctx["engine"]) @@ -553,13 +514,7 @@ def _full_graph(eng: _Engine) -> bool: if len(qs) != 1: return False q = qs.pop() - compilation = eng.recipe["compilation_config"] - mode = str(compilation.get("cudagraph_mode", "FULL_AND_PIECEWISE")).upper() - if compilation.get("mode") in (0, "NONE") and "cudagraph_mode" not in compilation or mode in ("NONE", "PIECEWISE"): - return False - sizes = compilation.get("cudagraph_capture_sizes") - top = max(sizes) if sizes else compilation.get("max_cudagraph_capture_size") or vllm_linear._capture_sizes()[-1] - if q * len(eng.reqs) > top: + if not eng.capture or q * len(eng.reqs) > eng.capture[-1]: return False need = AttentionCGSupport.UNIFORM_SINGLE_TOKEN_DECODE if q == 1 else AttentionCGSupport.UNIFORM_BATCH for gs in eng.runner.attn_groups: @@ -571,9 +526,8 @@ def _full_graph(eng: _Engine) -> bool: def _ready_upstream(ctx: dict) -> None: - """In serving, a layer that reuses an earlier layer's top-k waits on events that - earlier layer recorded in the same graph. Captured alone, it would wait on events - recorded outside the capture; record them here, already satisfied, instead.""" + """A top-k reusing layer waits on events its source layer records in the same graph; + captured alone, record them here (already satisfied) instead.""" stream = torch.cuda.current_stream() for m in ctx["engine"].runner.model.modules(): group = getattr(getattr(m, "impl", None), "index_group", None) or getattr(m, "index_group", None) @@ -584,16 +538,19 @@ def _ready_upstream(ctx: dict) -> None: def _launcher(ctx: dict): + from vllm.distributed.parallel_state import graph_capture eager = (lambda: _kernel(ctx)), False if not _cudagraph(ctx): return eager + err = None try: with _replaying(ctx): for _ in range(2): ctx["forward"](*ctx["args"], **ctx["kwargs"]) torch.cuda.synchronize() g = torch.cuda.CUDAGraph() - with torch.cuda.graph(g, pool=torch.cuda.graph_pool_handle()): + with graph_capture(torch.device("cuda", torch.cuda.current_device())) as gc_, \ + torch.cuda.graph(g, pool=torch.cuda.graph_pool_handle(), stream=gc_.stream): _ready_upstream(ctx) ctx["graph_out"] = ctx["forward"](*ctx["args"], **ctx["kwargs"]) torch.cuda.synchronize() @@ -601,7 +558,10 @@ def _launcher(ctx: dict): if vllm_linear._is_fault(e): raise torch.cuda.synchronize() - print(f"[vllm.attention] CUDA-graph capture failed, timing eagerly: {type(e).__name__}: {e}"[:300], + err = e + if not ranks.agree(err is None): # every rank replays, or every rank runs eagerly + print(f"[vllm.attention] CUDA-graph capture failed, timing eagerly: {type(err).__name__}: {err}"[:300] + if err else "[vllm.attention] another rank's CUDA-graph capture failed, timing eagerly", file=sys.stderr) return eager ctx["graph"] = g diff --git a/operatorx/runners/common/vllm/engine.py b/operatorx/runners/common/vllm/engine.py new file mode 100644 index 0000000000..7c5209067f --- /dev/null +++ b/operatorx/runners/common/vllm/engine.py @@ -0,0 +1,135 @@ +"""The vLLM engine a case runs under, configured as `vllm serve` configures it. + +engine_args(): the InferenceX recipe's serve arguments (OPERATORX_ENGINE_ARGS, less those +recipes.serve_argv does not apply) through vLLM's own parser, the op's parallel split and +the backend's overrides, with the external_launcher executor (one srun task per rank, +joined through env://). Layer ops enter that config once per process (context()); +attention builds an LLM from it. +""" +from __future__ import annotations + +import dataclasses +import json +import os +import tempfile + +import torch + +from operatorx.core import UnsupportedOpError, parallel + +# for layer cases without a checkpoint config; MoE, so vLLM builds the expert-parallel group +_STAND_IN = {"architectures": ["MixtralForCausalLM"], "model_type": "mixtral", "hidden_size": 4096, + "intermediate_size": 512, "num_attention_heads": 64, "num_key_value_heads": 8, + "num_hidden_layers": 1, "num_local_experts": 8, "num_experts_per_tok": 2, + "vocab_size": 1024, "max_position_embeddings": 65536, "torch_dtype": "bfloat16"} + +_STATE: dict | None = None +_META: dict = {} + + +def launch() -> None: + """The env:// rendezvous external_launcher reads (one device by default).""" + os.environ.setdefault("RANK", "0") + os.environ.setdefault("LOCAL_RANK", "0") + os.environ.setdefault("WORLD_SIZE", "1") + os.environ.setdefault("MASTER_ADDR", "127.0.0.1") + os.environ.setdefault("MASTER_PORT", str(29500 + os.getpid() % 1000)) + torch.cuda.set_device(int(os.environ["LOCAL_RANK"])) + + +def recipe_kwargs(without: tuple = ()) -> tuple[dict, dict]: + """The recipe's server arguments as EngineArgs kwargs (non-default fields only), and the + attention and compilation configs apart from them: attention is merged with the op's, and + an eager engine applies only the compilation config's custom-op selection.""" + from vllm.engine.arg_utils import EngineArgs + from vllm.utils.argparse_utils import FlexibleArgumentParser + + from operatorx.recipes import serve_argv + args = {k: v for k, v in json.loads(os.environ.get("OPERATORX_ENGINE_ARGS") or "{}").items() + if k not in without} + attention = json.loads(args.pop("attention-config", None) or "{}") + compilation = json.loads(args.pop("compilation-config", None) or "{}") + if args.get("max-cudagraph-capture-size"): + compilation.setdefault("max_cudagraph_capture_size", int(args["max-cudagraph-capture-size"])) + info = {"attention_config": attention, "compilation_config": compilation} + if not args: + return {}, info + parser = EngineArgs.add_cli_args(FlexibleArgumentParser()) + base = vars(parser.parse_known_args(["--model", "m"])[0]) + ns, frontend_only = parser.parse_known_args(["--model", "m", *serve_argv(args)]) + fields = {f.name for f in dataclasses.fields(EngineArgs)} - {"model"} + kwargs = {k: v for k, v in vars(ns).items() if k in fields and v != base.get(k)} + info["ignored_args"] = frontend_only + return kwargs, info + + +def engine_args(split: dict | None, without: tuple = (), eager: bool = False, defaults: dict | None = None, + **overrides) -> tuple[dict, dict]: + """EngineArgs kwargs for a case: defaults, the recipe's, the split, the overrides (an + attention_config override merges into the recipe's); plus the recipe's attention and + compilation configs.""" + kw, info = recipe_kwargs(without) + kw = {**(defaults or {}), **kw} + compilation = info["compilation_config"] + if not eager and compilation: + kw["compilation_config"] = compilation + elif compilation.get("custom_ops"): # custom-op selection holds without compilation + kw["compilation_config"] = {"custom_ops": compilation["custom_ops"]} + key = parallel.normalize(split) + kw.update(tensor_parallel_size=key["tp"], data_parallel_size=key["dp"], + decode_context_parallel_size=key["dcp"], enable_expert_parallel=key["ep"] > 1, + distributed_executor_backend="external_launcher") + attention = {**info["attention_config"], **overrides.pop("attention_config", {})} + kw.update(overrides) + if attention: + kw["attention_config"] = attention + _META.update(engine_args=kw, ignored_args=info.get("ignored_args", [])) + return kw, info + + +def full_graph_sizes(kw: dict, compilation: dict) -> list[int]: + """The sizes vLLM captures full CUDA graphs at for these arguments and the recipe's + compilation config, as a non-eager engine would; none when it captures no full graphs.""" + from vllm.engine.arg_utils import EngineArgs + cfg = EngineArgs(**{**kw, "enforce_eager": False, "compilation_config": compilation}).create_engine_config() + cc = cfg.compilation_config + return sorted(cc.cudagraph_capture_sizes or []) if cc.cudagraph_mode.has_full_cudagraphs() else [] + + +def _model_dir() -> str: + path = os.environ.get("OPERATORX_MODEL_CONFIG") + if path: + return path + path = tempfile.mkdtemp(prefix="opx_vllm_cfg_") + with open(os.path.join(path, "config.json"), "w") as f: + json.dump(_STAND_IN, f) + return path + + +def context(split: dict | None = None): + """Enter vLLM's config and the worker's distributed environment, once per process.""" + global _STATE + key = parallel.normalize(split) + if _STATE is not None: + if _STATE["split"] != key: + raise UnsupportedOpError(f"this process runs parallel={_STATE['split']}, not {key}") + return _STATE["config"] + from vllm.config import set_current_vllm_config + from vllm.engine.arg_utils import EngineArgs + from vllm.v1.worker.gpu_worker import init_worker_distributed_environment + from vllm.v1.worker.workspace import init_workspace_manager + + launch() + kw, _ = engine_args(split, model=_model_dir(), skip_tokenizer_init=True, load_format="dummy") + cfg = EngineArgs(**kw).create_engine_config() + ctx = set_current_vllm_config(cfg) + ctx.__enter__() + local_rank = int(os.environ["LOCAL_RANK"]) + init_worker_distributed_environment(cfg, int(os.environ["RANK"]), "env://", local_rank) + init_workspace_manager(torch.device("cuda", local_rank)) + _STATE = {"split": key, "config": cfg, "ctx": ctx} + return cfg + + +def meta() -> dict: + return {k: json.loads(json.dumps(v, default=str)) for k, v in _META.items()} diff --git a/operatorx/runners/common/vllm/linear.py b/operatorx/runners/common/vllm/linear.py index 3555e8360e..c8369b5a43 100644 --- a/operatorx/runners/common/vllm/linear.py +++ b/operatorx/runners/common/vllm/linear.py @@ -1,29 +1,25 @@ -"""Dense GEMM as a real vLLM linear layer, so vLLM picks the kernel. - -Each op builds a vLLM ReplicatedLinear under the quantization config its -weight scheme implies (the same config classes a checkpoint's -quantization_config selects), loads synthetic weights in checkpoint format, -runs vLLM's process_weights_after_loading, and times layer(x) on a bf16 -activation - so activation quantization, kernel selection and any weight -repacking are vLLM's own - replayed as a CUDA graph where vLLM would capture -one. The kernel vLLM chose is reported per op. +"""Dense GEMM as a vLLM RowParallelLinear (K split over tp, all-reduced), so vLLM picks the kernel. + +Built under the quantization config a checkpoint with the operands' scheme would declare, with +synthetic checkpoint-format weights and process_weights_after_loading; layer(x) on a bf16 +activation, as a CUDA graph where vLLM would capture one. The chosen kernel is recorded. """ from __future__ import annotations -import json import os import sys -import tempfile import torch from operatorx.core import BackendImpl, Op, UnsupportedOpError, lookup_versions +from operatorx.runners.common import ranks +from operatorx.runners.common.vllm import engine PER_TENSOR, PER_TOKEN, PER_CHANNEL = [-1, -1], [1, -1], [-1, 1] def device() -> "torch.device": - """The accelerator vLLM is running on (cuda on NVIDIA and ROCm).""" + """cuda on NVIDIA and ROCm.""" from vllm.platforms import current_platform return torch.device(current_platform.device_type) @@ -35,7 +31,7 @@ def sync() -> None: "round_method": "half_even", "scale_type": "float", "scale_format": "e8m0", "scale_calculation_mode": "even", "mx_element_dtype": None, "observer_cls": "PerBlockMXObserver", "is_scale_quant": False} -# The quantization_config AMD's MXFP4 checkpoints ship (e.g. amd/Kimi-K2.5-MXFP4), exclusions dropped. +# the quantization_config AMD's MXFP4 checkpoints ship (e.g. amd/Kimi-K2.5-MXFP4), exclusions dropped _QUARK_MXFP4 = { "quant_method": "quark", "quant_mode": "eager_mode", "exclude": [], "algo_config": None, "global_quant_config": {"input_tensors": {**_MX_FP4, "is_dynamic": True}, @@ -47,7 +43,7 @@ def sync() -> None: } -# Kimi-K3's routed experts: MXFP4 weights, bf16 activations (compressed-tensors mxfp4-pack). +# Kimi-K3's routed experts: MXFP4 weights, bf16 activations (compressed-tensors mxfp4-pack) _CT_MXFP4_W4A16 = { "quant_method": "compressed-tensors", "format": "mxfp4-pack-quantized", "ignore": [], "config_groups": {"group_0": { @@ -59,8 +55,7 @@ def sync() -> None: def _scheme(qa: dict, qb: dict) -> tuple[str, dict] | None: - """Operand descriptors -> (vLLM quant method, checkpoint quantization_config), as a - checkpoint with that scheme would declare it; None for unquantized.""" + """Operands -> (quant method, checkpoint quantization_config); None: unquantized; (): unsupported.""" if "scale" not in qa and "scale" not in qb: return None if qa["dtype"] == qb["dtype"] == "bf16" else () sa, sb = qa.get("scale"), qb.get("scale") @@ -83,7 +78,7 @@ def _scheme(qa: dict, qb: dict) -> tuple[str, dict] | None: if sb["dtype"] in ("fp32", "bf16"): # bf16 block scales load like fp32 ones return "fp8", cfg if sb["dtype"] == "ue8m0": - # DeepSeek-V4: vLLM remaps the checkpoint's fp8 config to deepseek_v4_fp8 (ue8m0 scales). + # DeepSeek-V4: vLLM remaps the checkpoint's fp8 config to deepseek_v4_fp8 return "deepseek_v4_fp8", {**cfg, "scale_fmt": "ue8m0"} if ga == PER_TOKEN and gb == PER_CHANNEL and not sa["static"]: return "compressed-tensors", { @@ -104,52 +99,14 @@ def _scheme(qa: dict, qb: dict) -> tuple[str, dict] | None: return () -# Env that steers vLLM's linear-kernel choice; recorded with every result. +# env that steers vLLM's linear-kernel choice; recorded with every result _ENV_KEYS = ("VLLM_USE_DEEP_GEMM", "VLLM_USE_DEEP_GEMM_E8M0", "VLLM_BLOCKSCALE_FP8_GEMM_FLASHINFER", "VLLM_ROCM_USE_AITER", "VLLM_ROCM_USE_AITER_LINEAR") -_READY = None - - def versions() -> dict[str, str]: return lookup_versions("vllm", "torch") -def _vllm_context(): - """Enter a minimal vLLM config + single-rank parallel state once per process.""" - global _READY - if _READY is not None: - return _READY - from vllm.config import ModelConfig, VllmConfig, set_current_vllm_config - from vllm.distributed import init_distributed_environment, initialize_model_parallel - - cfg_dir = tempfile.mkdtemp(prefix="opx_vllm_cfg_") - with open(os.path.join(cfg_dir, "config.json"), "w") as f: - json.dump({"architectures": ["LlamaForCausalLM"], "model_type": "llama", "hidden_size": 256, - "intermediate_size": 512, "num_attention_heads": 4, "num_key_value_heads": 4, - "num_hidden_layers": 1, "vocab_size": 1024, "max_position_embeddings": 2048, - "torch_dtype": "bfloat16"}, f) - vcfg = VllmConfig() - vcfg.model_config = ModelConfig(model=cfg_dir, dtype="bfloat16", skip_tokenizer_init=True) - try: - vcfg._set_cudagraph_sizes() # vLLM's default capture sizes for this config - except Exception: - pass - ctx = set_current_vllm_config(vcfg) - ctx.__enter__() - init_distributed_environment(world_size=1, rank=0, local_rank=torch.cuda.current_device(), - distributed_init_method=f"tcp://127.0.0.1:{29500 + os.getpid() % 1000}", - backend="nccl") - # with no model config vLLM also builds the expert-parallel group (size 1), which MoE layers need - model_config, vcfg.model_config = vcfg.model_config, None - try: - initialize_model_parallel(1, 1) - finally: - vcfg.model_config = model_config - _READY = ctx - return ctx - - def _quant_config(args): sch = _scheme(args["a"], args["b"]) if sch is None: @@ -160,7 +117,6 @@ def _quant_config(args): def quant_config(method: str, cfg: dict): - """Instantiate the config vLLM registers for a checkpoint's quant_method.""" from vllm.model_executor.layers.quantization import QUANTIZATION_METHODS, get_quantization_config if method not in QUANTIZATION_METHODS: raise UnsupportedOpError(f"vLLM has no {method!r} quantization method") @@ -168,10 +124,8 @@ def quant_config(method: str, cfg: dict): def _reject_fallback(qc, method) -> None: - """A quantized checkpoint whose layer got an unquantized method computed something - else - bf16 weights and a bf16 matmul - and would be recorded under the quantized - row it is not. vLLM's mxfp4 does this for linear layers, where only its MoE experts - have a kernel.""" + """An unquantized fallback method would record a bf16 matmul under the quantized row + (vLLM's mxfp4 does this for linear layers; only its MoE experts have a kernel).""" if qc is not None and "Unquantized" in type(method).__name__: raise UnsupportedOpError( f"{type(qc).__name__} has no quantized linear method here; the layer fell back " @@ -179,8 +133,8 @@ def _reject_fallback(qc, method) -> None: def _set_quant_fp8_op(qc) -> None: - """VllmConfig turns on the CUDA quant_fp8 custom op for checkpoints with blocked - weights; mirror that per layer (CustomOp reads it when the layer is built).""" + """VllmConfig enables +quant_fp8 for blocked-weight checkpoints; mirror it per layer + (CustomOp reads it at build).""" from vllm.config import get_current_vllm_config ops = get_current_vllm_config().compilation_config.custom_ops blocked = getattr(qc, "weight_block_size", None) is not None @@ -198,7 +152,6 @@ def _is_fault(e: BaseException) -> bool: def _fill(layer: torch.nn.Module) -> None: - """Synthetic checkpoint-format values for whatever parameters the method registered.""" for name, p in layer.named_parameters(recurse=False): with torch.no_grad(): if p.dtype in (torch.float8_e4m3fn, torch.float8_e5m2, torch.float8_e4m3fnuz): @@ -212,7 +165,6 @@ def _fill(layer: torch.nn.Module) -> None: def _kernel_names(layer) -> dict[str, str]: - """Kernel objects vLLM attached to the quant method or its scheme.""" qm = layer.quant_method out = {} for owner in (qm, getattr(qm, "scheme", None), getattr(layer, "scheme", None)): @@ -228,7 +180,6 @@ def _kernel_names(layer) -> dict[str, str]: def _param_dtypes(layer) -> dict[str, str]: - """Post-processing dtype of every weight/scale tensor, as the kernel sees it.""" return {n: str(p.dtype).removeprefix("torch.") for n, p in layer.named_parameters(recurse=False)} @@ -245,53 +196,54 @@ def _prepare_gemm(op: Op) -> dict: return _prepare_fp32_out(op) if a.get("out", "bf16") != "bf16": raise UnsupportedOpError(f"vLLM linear layers return bf16 or fp32; got {a['out']!r}") - _vllm_context() + tp = (a.get("parallel") or {}).get("tp", 1) + if k % tp: + raise UnsupportedOpError(f"k={k} does not split {tp} ways") + engine.context(a.get("parallel")) import vllm.envs as envs - from vllm.model_executor.layers.linear import ReplicatedLinear - qc = _quant_config(a) - _set_quant_fp8_op(qc) - prev = torch.get_default_dtype() - torch.set_default_dtype(torch.bfloat16) # as vLLM's model loader does while building layers - try: - layer = ReplicatedLinear(k, n, bias=bool(a.get("bias")), quant_config=qc, params_dtype=torch.bfloat16, - prefix="model.layers.0.mlp.down_proj", disable_tp=True).to(device()) - # set by the column/row-parallel linears real layers use; some kernels read them - for attr, v in (("input_size_per_partition", k), ("output_size_per_partition", n)): - if not hasattr(layer, attr): - setattr(layer, attr, v) - _fill(layer) - loaded = _param_dtypes(layer) - qm = layer.quant_method - _reject_fallback(qc, qm) - if hasattr(qm, "process_weights_after_loading"): - qm.process_weights_after_loading(layer) - except (NotImplementedError, AssertionError, ValueError, RuntimeError) as e: - if _is_fault(e): - raise - raise UnsupportedOpError(f"vLLM rejected {a}: {type(e).__name__}: {e}"[:400]) from e - finally: - torch.set_default_dtype(prev) - kernels = _kernel_names(layer) - if any(v.startswith("Emulation") for v in kernels.values()): - raise UnsupportedOpError(f"vLLM has only an emulation kernel for {a} here: {kernels}") - x = torch.randn(m, k, device=device(), dtype=torch.bfloat16) + from vllm.model_executor.layers.linear import RowParallelLinear + with ranks.together("build this layer"): + qc = _quant_config(a) + _set_quant_fp8_op(qc) + prev = torch.get_default_dtype() + torch.set_default_dtype(torch.bfloat16) # as vLLM's model loader does while building layers + try: + layer = RowParallelLinear(k, n, bias=bool(a.get("bias")), quant_config=qc, params_dtype=torch.bfloat16, + reduce_results=True, prefix="model.layers.0.mlp.down_proj").to(device()) + _fill(layer) + loaded = _param_dtypes(layer) + qm = layer.quant_method + _reject_fallback(qc, qm) + if hasattr(qm, "process_weights_after_loading"): + qm.process_weights_after_loading(layer) + except (NotImplementedError, AssertionError, ValueError, RuntimeError) as e: + if _is_fault(e): + raise + raise UnsupportedOpError(f"vLLM rejected {a}: {type(e).__name__}: {e}"[:400]) from e + finally: + torch.set_default_dtype(prev) + kernels = _kernel_names(layer) + if any(v.startswith("Emulation") for v in kernels.values()): + raise UnsupportedOpError(f"vLLM has only an emulation kernel for {a} here: {kernels}") + x = torch.randn(m, layer.input_size_per_partition, device=device(), dtype=torch.bfloat16) ctx = {"layer": layer, "x": x, "meta": {"vllm_quant_method": type(qm).__name__, "vllm_kernels": kernels, "param_dtypes_loaded": loaded, "param_dtypes": _param_dtypes(layer), - "vllm_env": {k: getattr(envs, k) for k in _ENV_KEYS if hasattr(envs, k)}}} - try: - _kernel_gemm(ctx) - sync() - except (NotImplementedError, AssertionError, RuntimeError, ValueError) as e: - if _is_fault(e): - raise - raise UnsupportedOpError(f"vLLM kernel failed for {a}: {type(e).__name__}: {e}"[:400]) from e + "vllm_env": {k: getattr(envs, k) for k in _ENV_KEYS if hasattr(envs, k)}, + "vllm_engine": engine.meta()}} + with ranks.together("run this layer"): + try: + _kernel_gemm(ctx) + sync() + except (NotImplementedError, AssertionError, RuntimeError, ValueError) as e: + if _is_fault(e): + raise + raise UnsupportedOpError(f"vLLM kernel failed for {a}: {type(e).__name__}: {e}"[:400]) from e return ctx class _TorchMM(torch.nn.Module): - """The matmul vLLM runs for fp32-output gates and routers (GateLinear's cuBLAS tier): - torch.mm with an fp32 epilogue on a bf16 weight; an fp32 matmul on an fp32 weight.""" + """vLLM's fp32-output gate/router matmul (GateLinear's cuBLAS tier).""" def __init__(self, k: int, n: int, dtype: torch.dtype, bias: bool): super().__init__() @@ -312,7 +264,9 @@ def _prepare_fp32_out(op: Op) -> dict: if "scale" in qa or "scale" in qb or qa["dtype"] != "bf16" or qb["dtype"] not in ("bf16", "fp32") \ or a.get("out", "bf16") != "fp32": raise UnsupportedOpError(f"no fp32-output vLLM matmul for a={qa} b={qb} out={a.get('out')}") - _vllm_context() + if (a.get("parallel") or {}).get("tp", 1) > 1: + raise UnsupportedOpError("vLLM's fp32-output gates and routers are replicated, never split") + engine.context(a.get("parallel")) layer = _TorchMM(a["k"], a["n"], getattr(torch, {"bf16": "bfloat16", "fp32": "float32"}[qb["dtype"]]), bool(a.get("bias"))) ctx = {"layer": layer, "x": torch.randn(a["m"], a["k"], device=device(), dtype=torch.bfloat16), @@ -329,42 +283,39 @@ def _kernel_gemm(ctx: dict) -> None: def _capture_sizes() -> list[int]: from vllm.config import get_current_vllm_config - sizes = get_current_vllm_config().compilation_config.cudagraph_capture_sizes - if sizes: - return sorted(sizes) - # VllmConfig._set_cudagraph_sizes' default candidates - from vllm.platforms import current_platform - top = 1024 if current_platform.is_device_capability_family(100) else 512 - return [1, 2, 4] + list(range(8, 256, 8)) + list(range(256, top + 1, 16)) + return sorted(get_current_vllm_config().compilation_config.cudagraph_capture_sizes or []) def _launcher(ctx: dict): - """(callable to time, whether it is a CUDA-graph replay). As vLLM serves it, a batch - of up to the max capture size replays a graph captured at the next capture size - (the batch padded up to it); larger batches run eagerly. The cap is vLLM's default - unless OPERATORX_CUDA_GRAPH_MAX_TOKENS sets it (0 = never graph); deployments set - their own (max-cudagraph-capture-size, cudagraph_mode).""" + """(callable, cuda_graph): as served, a batch up to the largest capture size replays the next + size's graph, larger ones run eagerly. OPERATORX_CUDA_GRAPH_MAX_TOKENS caps it (0 = never).""" + from vllm.distributed.parallel_state import graph_capture x = ctx["x"] eager = (lambda: _kernel_gemm(ctx)), False # noqa: E731 sizes = _capture_sizes() - cap = int(os.environ.get("OPERATORX_CUDA_GRAPH_MAX_TOKENS", sizes[-1])) + cap = int(os.environ.get("OPERATORX_CUDA_GRAPH_MAX_TOKENS", sizes[-1] if sizes else 0)) size = next((s for s in sizes if s >= x.shape[0]), None) if cap <= 0 or size is None or size > cap: return eager - xp = torch.randn(size, x.shape[1], device=x.device, dtype=x.dtype) + xp = torch.randn(size, *x.shape[1:], device=x.device, dtype=x.dtype) + err = None try: for _ in range(2): ctx["layer"](xp) torch.cuda.synchronize() g = torch.cuda.CUDAGraph() - with torch.cuda.graph(g, pool=torch.cuda.graph_pool_handle()): + with graph_capture(x.device) as gc, torch.cuda.graph(g, pool=torch.cuda.graph_pool_handle(), + stream=gc.stream): ctx["graph_out"] = ctx["layer"](xp) torch.cuda.synchronize() except Exception as e: if _is_fault(e): raise torch.cuda.synchronize() - print(f"[vllm.linear] CUDA-graph capture failed, timing eagerly: {type(e).__name__}: {e}"[:300], + err = e + if not ranks.agree(err is None): # every rank replays, or every rank runs eagerly + print(f"[vllm.linear] CUDA-graph capture failed, timing eagerly: {type(err).__name__}: {err}"[:300] + if err else "[vllm.linear] another rank's CUDA-graph capture failed, timing eagerly", file=sys.stderr) return eager ctx["graph"], ctx["x_padded"] = g, xp diff --git a/operatorx/runners/common/vllm/modules.py b/operatorx/runners/common/vllm/modules.py new file mode 100644 index 0000000000..b3326a9678 --- /dev/null +++ b/operatorx/runners/common/vllm/modules.py @@ -0,0 +1,72 @@ +"""A checkpoint's own vLLM MoE module, built as its decoder layer builds it, with dummy +weights and post-load processing (opt-in: OPERATORX_MODEL_MODULES=1).""" +from __future__ import annotations + +import importlib + +import torch + + +def _layer(pkg: str): + from vllm.platforms import current_platform + return importlib.import_module(f"vllm.models.{pkg}.{'amd' if current_platform.is_rocm() else 'nvidia'}.model") + + +def _deepseek_v4(pkg): + def build(cfg, hf, op_args): + m = _layer(pkg) + hash_layers = getattr(hf, "num_hash_layers", None) or 0 + i = 0 if op_args["router"]["select"]["kind"] == "hash" else hash_layers + kw = {"num_hash_layers": hash_layers} if pkg == "deepseek_v4" else {} + return m.DeepseekV4MoE(cfg, prefix=f"model.layers.{i}.ffn", + use_sequence_parallel=m._use_sequence_parallel(cfg), **kw) + return build + + +def _minimax_m3(cfg, hf, op_args): + m = _layer("minimax_m3") + i = next(i for i in range(hf.num_hidden_layers) if m._is_moe_layer(hf, i)) + # reduce here: the op includes the reduction the model fuses into the next RMSNorm + return m.MiniMaxM3MoE(config=hf, layer_id=i, quant_config=cfg.quant_config, reduce_results=True, + prefix=f"model.layers.{i}.block_sparse_moe") + + +def _kimi_k3(cfg, hf, op_args): + m = _layer("kimi_k3") + i = hf.first_k_dense_replace + p = cfg.parallel_config + use_sp = (p.pipeline_parallel_size == 1 and p.enable_expert_parallel and p.tensor_parallel_size > 1 + and (cfg.kernel_config.moe_backend == "deep_gemm_mega_moe" or p.data_parallel_size > 1)) + return m.KimiMoE(config=hf, vllm_config=cfg, quant_config=cfg.quant_config, + prefix=f"model.layers.{i}.block_sparse_moe", layer_idx=i, use_sequence_parallel=use_sp) + + +BUILDERS = { + "DeepseekV4ForCausalLM": _deepseek_v4("deepseek_v4"), + "DeepseekV41ForCausalLM": _deepseek_v4("deepseek_v41"), + "MiniMaxM3SparseForConditionalGeneration": _minimax_m3, + "KimiK3ForConditionalGeneration": _kimi_k3, +} + + +def build(op_args: dict) -> torch.nn.Module | None: + """None when the engine's model is a stand-in or has no entry here.""" + from vllm.config import get_current_vllm_config + from vllm.model_executor.model_loader import get_model_loader + from vllm.model_executor.model_loader.utils import process_weights_after_loading + + cfg = get_current_vllm_config() + arch = next((a for a in (cfg.model_config.architectures or ()) if a in BUILDERS), None) + if arch is None: + return None + device = torch.device("cuda", torch.cuda.current_device()) + prev = torch.get_default_dtype() + torch.set_default_dtype(cfg.model_config.dtype) + try: + with device: + module = BUILDERS[arch](cfg, cfg.model_config.hf_text_config, op_args) + finally: + torch.set_default_dtype(prev) + get_model_loader(cfg.load_config).load_weights(module, cfg.model_config) + process_weights_after_loading(module, cfg.model_config, device) + return module diff --git a/operatorx/runners/common/vllm/moe.py b/operatorx/runners/common/vllm/moe.py index 1f9dfc3e18..50dce271db 100644 --- a/operatorx/runners/common/vllm/moe.py +++ b/operatorx/runners/common/vllm/moe.py @@ -1,18 +1,19 @@ -"""MoE layer through vLLM's own MoE pipeline, so vLLM picks router, expert and -shared-expert kernels and the stream layout. - -Each op builds the pieces a vLLM MoE block wires together - a GateLinear -router, the routed experts from FusedMoEFactory under the quantization config -the expert descriptors imply, and a shared-expert MLP under its own - loads -synthetic checkpoint-format weights, runs process_weights_after_loading and -times the block on bf16 hidden states (as a CUDA graph where vLLM would -capture one). The quant methods and expert kernels vLLM chose are reported. +"""MoE layer through vLLM's MoE pipeline, so vLLM picks router, expert and shared-expert kernels. + +A GateLinear router, FusedMoEFactory experts and a shared-expert MLP under the quant configs the +descriptors imply, synthetic weights + process_weights_after_loading, timed on bf16 x (CUDA graph +where vLLM would capture one). Splits shard as vLLM's parallel config does; shared experts and +sequence-parallel chunking follow DeepseekV2MLP / DeepseekV2MoE. """ from __future__ import annotations +import os + import torch from operatorx.core import BackendImpl, Op, UnsupportedOpError +from operatorx.runners.common import ranks +from operatorx.runners.common.vllm import engine, modules from operatorx.runners.common.vllm import linear as vllm_linear from operatorx.runners.common.vllm.linear import (_ENV_KEYS, _fill, _is_fault, _launcher, device, sync, versions) @@ -21,19 +22,9 @@ _DTYPES = {"bf16": torch.bfloat16, "fp32": torch.float32} _MAX_TOKENS = 65536 # the largest token count in the testlists -_WORKSPACE = False _LAYER = 0 -def _context(): - global _WORKSPACE - vllm_linear._vllm_context() - if not _WORKSPACE: - from vllm.v1.worker.workspace import init_workspace_manager - init_workspace_manager(torch.device("cuda", torch.cuda.current_device())) - _WORKSPACE = True - - def _quant(x: dict, w: dict, where: str): """(activation, weight) descriptors -> vLLM quantization config, or None for bf16.""" if x.get("input", "bf16") != "bf16": @@ -77,15 +68,29 @@ def _act_fn(act: dict): raise UnsupportedOpError(f"shared-expert activation {kind!r} is not wired") +def _dp_tokens(cache: dict, T: int): + """DP ranks' token counts, coordinated as the model runner does per step (untimed).""" + if T not in cache: + from vllm.config import get_current_vllm_config + from vllm.v1.worker.dp_utils import coordinate_batch_across_dp + cfg = get_current_vllm_config() + graph = int(bool(cfg.compilation_config.cudagraph_capture_sizes)) + cache[T] = coordinate_batch_across_dp(T, False, cfg.parallel_config, cudagraph_mode=graph)[1] + return cache[T] + + class _SharedMLP(torch.nn.Module): - def __init__(self, hidden: int, inter: int, act: dict, qc, prefix: str, expert_gate=None): + + def __init__(self, hidden: int, inter: int, act: dict, qc, prefix: str, expert_gate=None, + sequence_parallel: bool = False): super().__init__() self.expert_gate = expert_gate # Qwen: sigmoid(gate(x)) scales the shared output from vllm.model_executor.layers.linear import MergedColumnParallelLinear, RowParallelLinear self.gate_up_proj = MergedColumnParallelLinear(hidden, [inter] * 2, bias=False, quant_config=qc, - disable_tp=True, prefix=f"{prefix}.gate_up_proj") + disable_tp=sequence_parallel, + prefix=f"{prefix}.gate_up_proj") self.down_proj = RowParallelLinear(inter, hidden, bias=False, quant_config=qc, reduce_results=False, - disable_tp=True, prefix=f"{prefix}.down_proj") + disable_tp=sequence_parallel, prefix=f"{prefix}.down_proj") self.act_fn = _act_fn(act) def forward(self, x): @@ -96,8 +101,8 @@ def forward(self, x): return h -# DeepSeek-V4 routed experts: MXFP4 weights under the checkpoint's fp8 (ue8m0) config, -# which vLLM routes to its MXFP4 MoE method (expert_dtype "fp4"). +# DeepSeek-V4 routed experts: MXFP4 weights under the checkpoint's fp8 (ue8m0) config, which vLLM +# routes to its MXFP4 MoE method (expert_dtype "fp4") _DSV4_FP4_EXPERTS = ("deepseek_v4_fp8", {"quant_method": "fp8", "activation_scheme": "dynamic", "fmt": "e4m3", "scale_fmt": "ue8m0", "weight_block_size": [128, 128]}) @@ -165,6 +170,7 @@ class _MoeBlock(torch.nn.Module): def __init__(self, a: dict, prefix: str): super().__init__() + from vllm.config import get_current_vllm_config from vllm.model_executor.layers.fused_moe.layer import FusedMoEFactory from vllm.model_executor.layers.fused_moe.router.gate_linear import GateLinear from vllm.model_executor.layers.fused_moe.utils import resolve_layer_fused_shared_expert @@ -178,6 +184,8 @@ def __init__(self, a: dict, prefix: str): qc = _expert_quant(ex["a1"], ex["w1"]) vllm_linear._set_quant_fp8_op(qc) H, E, K = a["hidden"], ex["num"], ex["top_k"] + self.sequence_parallel = get_current_vllm_config().parallel_config.use_sequence_parallel_moe + self._dp_tokens: dict[int, torch.Tensor | None] = {} L = ex.get("latent") or H sel = rt["select"] routing = a.get("routing") or {"distribution": "natural", "seed": 0} @@ -216,7 +224,7 @@ def __init__(self, a: dict, prefix: str): shared_gate = expert_gate else: shared = _SharedMLP(H, sh["inter"] * sh["count"], act, sqc, f"{prefix}.shared_experts", - expert_gate=expert_gate) + expert_gate=expert_gate, sequence_parallel=self.sequence_parallel) latent = {} self.down_proj = None if ex.get("latent"): # Kimi-K3: routed experts run at width L between bf16 projections @@ -242,7 +250,7 @@ def __init__(self, a: dict, prefix: str): routed_scaling_factor=rt.get("scale") or 1.0, e_score_correction_bias=bias, hash_indices_table=self.gate.tid2eid, custom_routing_function=custom, - has_bias=bool(ex.get("bias")), reduce_results=False, + has_bias=bool(ex.get("bias")), reduce_results=True, is_sequence_parallel=self.sequence_parallel, n_shared_experts=sh["count"] if fused_shared else None, fuse_shared_experts=fused_shared, router_logits_dtype=self.gate.out_dtype, gate=None if ex.get("latent") else self.gate, shared_experts=shared, shared_expert_gate=shared_gate, **latent, **_act_kwargs(act)) @@ -253,24 +261,66 @@ def __init__(self, a: dict, prefix: str): self._aux = aux_stream() self._events = (torch.cuda.Event(), torch.cuda.Event()) + def dp_tokens(self, T: int): + return _dp_tokens(self._dp_tokens, T) + def forward(self, x): from vllm.config import get_current_vllm_config from vllm.forward_context import set_forward_context T = x.shape[0] - with set_forward_context(None, get_current_vllm_config(), num_tokens=T): - if self.down_proj is None or not hasattr(self, "_aux"): - if self.down_proj is not None: # latent projection without an aux stream - lat, _ = self.down_proj(x) - return self.experts(hidden_states=lat, router_logits=self.gate(x)[0], - shared_experts_input=x) - ids = None if self.input_ids is None else self.input_ids[:T] - return self.experts(hidden_states=x, router_logits=x, input_ids=ids) - # as Kimi-K3's block: router and latent down projection on two streams at decode sizes - from vllm.utils.multi_stream_utils import maybe_execute_in_parallel - logits, (lat, _) = maybe_execute_in_parallel( - lambda: self.gate(x)[0], lambda: self.down_proj(x), self._events[0], self._events[1], - self._aux if T <= self._stream_tokens else None) - return self.experts(hidden_states=lat, router_logits=logits, shared_experts_input=x) + with set_forward_context(None, get_current_vllm_config(), num_tokens=T, + num_tokens_across_dp=self.dp_tokens(T)): + if not self.sequence_parallel: + return self._experts(x) + from vllm.distributed import tensor_model_parallel_all_gather + from vllm.model_executor.models.utils import sequence_parallel_chunk + return tensor_model_parallel_all_gather(self._experts(sequence_parallel_chunk(x)), 0)[:T] + + def _experts(self, x): + T = x.shape[0] + if self.down_proj is None or not hasattr(self, "_aux"): + if self.down_proj is not None: # latent projection without an aux stream + lat, _ = self.down_proj(x) + return self.experts(hidden_states=lat, router_logits=self.gate(x)[0], + shared_experts_input=x) + ids = None if self.input_ids is None else self.input_ids[:T] + return self.experts(hidden_states=x, router_logits=x, input_ids=ids) + # as Kimi-K3's block: router and latent down projection on two streams at decode sizes + from vllm.utils.multi_stream_utils import maybe_execute_in_parallel + logits, (lat, _) = maybe_execute_in_parallel( + lambda: self.gate(x)[0], lambda: self.down_proj(x), self._events[0], self._events[1], + self._aux if T <= self._stream_tokens else None) + return self.experts(hidden_states=lat, router_logits=logits, shared_experts_input=x) + + +class _ModelBlock(torch.nn.Module): + """A checkpoint's own MoE module (modules.py), called as its decoder layer calls it.""" + + def __init__(self, module: torch.nn.Module): + super().__init__() + import inspect + + from vllm.config import get_current_vllm_config + self.module = module + self.sequence_parallel = False # handled inside the module + self.fused_shared = None + self._dp_tokens: dict[int, torch.Tensor | None] = {} + self.takes_ids = "input_ids" in inspect.signature(module.forward).parameters + vocab = get_current_vllm_config().model_config.hf_text_config.vocab_size + self.register_buffer("input_ids", torch.randint(0, vocab, (_MAX_TOKENS,), device=device())) + + def dp_tokens(self, T: int): + return _dp_tokens(self._dp_tokens, T) + + def forward(self, x): + from vllm.config import get_current_vllm_config + from vllm.forward_context import set_forward_context + T = x.shape[0] + with set_forward_context(None, get_current_vllm_config(), num_tokens=T, + num_tokens_across_dp=self.dp_tokens(T)): + if self.takes_ids: + return self.module(x, input_ids=self.input_ids[:T]) + return self.module(x) def _describe(block) -> dict: @@ -295,8 +345,7 @@ def _describe(block) -> dict: def _release_previous() -> None: - """vLLM registers every MoE layer by name in the forward context; drop the previous - op's so its weights and workspaces are freed before the next layer is built.""" + """Drop the previous op's layers from vLLM's forward context so their memory is freed.""" import gc from vllm.config import get_current_vllm_config @@ -312,6 +361,29 @@ def _missing_op(e: BaseException) -> bool: return isinstance(e, AttributeError) and "_OpNamespace" in str(e) +def _generic_block(a: dict, prefix: str) -> _MoeBlock: + from vllm.model_executor.layers.quantization.base_config import QuantizeMethodBase + block = _MoeBlock(a, prefix).to(device()) + for m in block.modules(): + _fill(m) + if block.gate.e_score_correction_bias is not None: + block.gate.e_score_correction_bias.data.zero_() + if block.gate.tid2eid is not None: # the filler knows nothing of expert ids + block.gate.tid2eid.data.random_(0, a["experts"]["num"]) + methods = [] + for m in block.modules(): + qm = getattr(m, "quant_method", None) + if isinstance(qm, QuantizeMethodBase): + methods.append(type(qm).__name__) + qm.process_weights_after_loading(m) + # an all-unquantized block computed bf16 under a quantized row; a bf16 shared expert alone is normal + if block.experts_quant_config is not None and methods and all("Unquantized" in n for n in methods): + raise UnsupportedOpError( + f"{type(block.experts_quant_config).__name__} has no quantized MoE method here; " + f"the layer fell back to {sorted(set(methods))}") + return block + + def _prepare_moe(op: Op) -> dict: global _LAYER a = op.args @@ -320,61 +392,50 @@ def _prepare_moe(op: Op) -> dict: if a["tokens"] > _MAX_TOKENS: raise UnsupportedOpError(f"tokens > {_MAX_TOKENS} is not wired") routing = a.get("routing") or {"distribution": "natural", "seed": 0} - _context() + engine.context(a.get("parallel")) _release_previous() import vllm.envs as envs - from vllm.model_executor.layers.quantization.base_config import QuantizeMethodBase torch.manual_seed(routing["seed"]) _LAYER += 1 # vLLM registers layers by name; each op builds a fresh one prefix = f"model.layers.{_LAYER}.mlp" - prev = torch.get_default_dtype() - torch.set_default_dtype(torch.bfloat16) - try: - block = _MoeBlock(a, prefix).to(device()) - for m in block.modules(): - _fill(m) - if block.gate.e_score_correction_bias is not None: - block.gate.e_score_correction_bias.data.zero_() - if block.gate.tid2eid is not None: # the filler knows nothing of expert ids - block.gate.tid2eid.data.random_(0, a["experts"]["num"]) - methods = [] - for m in block.modules(): - qm = getattr(m, "quant_method", None) - if isinstance(qm, QuantizeMethodBase): - methods.append(type(qm).__name__) - qm.process_weights_after_loading(m) - # A quantized block whose every method came out unquantized computed bf16 and - # would be recorded under the quantized row it is not. A bf16 shared expert - # beside quantized routed ones is normal, so only an all-unquantized block counts. - if block.experts_quant_config is not None and methods and all("Unquantized" in n for n in methods): - raise UnsupportedOpError( - f"{type(block.experts_quant_config).__name__} has no quantized MoE method here; " - f"the layer fell back to {sorted(set(methods))}") - except (NotImplementedError, AssertionError, ValueError, RuntimeError, TypeError, KeyError, - AttributeError) as e: - if _is_fault(e) or (isinstance(e, AttributeError) and not _missing_op(e)): - raise - raise UnsupportedOpError(f"vLLM rejected this MoE layer: {type(e).__name__}: {e}"[:400]) from e - finally: - torch.set_default_dtype(prev) - kernels = _describe(block) - if any("Emulation" in str(v) for e in kernels.values() for v in e.values()): - raise UnsupportedOpError(f"vLLM has only an emulation kernel for this MoE layer here: {kernels}") - x = torch.randn(a["tokens"], a["hidden"], device=device(), dtype=torch.bfloat16) + with ranks.together("build this layer"): + prev = torch.get_default_dtype() + torch.set_default_dtype(torch.bfloat16) + try: + own = modules.build(a) if os.environ.get("OPERATORX_MODEL_MODULES") == "1" else None + block = _ModelBlock(own) if own is not None else _generic_block(a, prefix) + except (NotImplementedError, AssertionError, ValueError, RuntimeError, TypeError, KeyError, + AttributeError) as e: + if _is_fault(e) or (isinstance(e, AttributeError) and not _missing_op(e)): + raise + raise UnsupportedOpError(f"vLLM rejected this MoE layer: {type(e).__name__}: {e}"[:400]) from e + finally: + torch.set_default_dtype(prev) + kernels = _describe(block) + if any("Emulation" in str(v) for e in kernels.values() for v in e.values()): + raise UnsupportedOpError(f"vLLM has only an emulation kernel for this MoE layer here: {kernels}") + # the tokens of one data-parallel group, identical on its tensor-parallel ranks + from vllm.distributed import get_dp_group + g = torch.Generator(device=device()).manual_seed(routing["seed"] * 1000 + get_dp_group().rank_in_group) + x = torch.randn(a["tokens"], a["hidden"], device=device(), dtype=torch.bfloat16, generator=g) ctx = {"layer": block, "x": x, "meta": {"vllm_modules": kernels, "fused_shared_experts": block.fused_shared, - "vllm_env": {k: getattr(envs, k) for k in (*_ENV_KEYS, *_MOE_ENV_KEYS) if hasattr(envs, k)}}} - try: - _kernel_moe(ctx) - sync() - except (NotImplementedError, AssertionError, RuntimeError, ValueError, TypeError, AttributeError) as e: - if _is_fault(e) or (isinstance(e, AttributeError) and not _missing_op(e)): - raise - raise UnsupportedOpError(f"vLLM MoE kernel failed: {type(e).__name__}: {e}"[:400]) from e + "sequence_parallel_moe": block.sequence_parallel, + "model_module": type(block.module).__name__ if isinstance(block, _ModelBlock) else None, + "vllm_env": {k: getattr(envs, k) for k in (*_ENV_KEYS, *_MOE_ENV_KEYS) if hasattr(envs, k)}, + "vllm_engine": engine.meta()}} + with ranks.together("run this layer"): + try: + _kernel_moe(ctx) + sync() + except (NotImplementedError, AssertionError, RuntimeError, ValueError, TypeError, AttributeError) as e: + if _is_fault(e) or (isinstance(e, AttributeError) and not _missing_op(e)): + raise + raise UnsupportedOpError(f"vLLM MoE kernel failed: {type(e).__name__}: {e}"[:400]) from e return ctx -# Env that steers vLLM's MoE kernel and stream choice; recorded with every result. +# env that steers vLLM's MoE kernel and stream choice; recorded with every result _MOE_ENV_KEYS = ("VLLM_ROCM_USE_AITER_MOE", "VLLM_ROCM_USE_AITER_FUSION_SHARED_EXPERTS", "VLLM_USE_FLASHINFER_MOE_FP8", "VLLM_USE_FLASHINFER_MOE_FP4", "VLLM_FLASHINFER_MOE_BACKEND", "VLLM_DISABLE_SHARED_EXPERTS_STREAM", "VLLM_SHARED_EXPERTS_STREAM_TOKEN_THRESHOLD") diff --git a/operatorx/runners/nvidia/backends/vllm.py b/operatorx/runners/nvidia/backends/vllm.py index 4ec34623ec..1e4bdf3da8 100644 --- a/operatorx/runners/nvidia/backends/vllm.py +++ b/operatorx/runners/nvidia/backends/vllm.py @@ -1,4 +1,4 @@ -"""Dense GEMM, MoE and attention modules through vLLM's own layers and kernel selection.""" +"""GEMM, MoE and attention through vLLM's layers and kernel selection.""" from operatorx.runners.common.vllm import attention, linear, moe from operatorx.runners.common.vllm.linear import versions diff --git a/operatorx/runners/nvidia/runner.py b/operatorx/runners/nvidia/runner.py index 803b8917e5..db3af54fcc 100644 --- a/operatorx/runners/nvidia/runner.py +++ b/operatorx/runners/nvidia/runner.py @@ -8,22 +8,8 @@ import torch from operatorx.core import BackendImpl, Op, Result, UnsupportedOpError -from operatorx.runners.common import profiling, telemetry +from operatorx.runners.common import profiling, ranks, telemetry, timing -_WARMUP = 5 -_ITERS = 10 -# Wall-clock warmup floor: iteration-count warmup alone is far shorter than -# the SM clock ramp after the inter-op cooldown. -_WARMUP_MIN_S = float(os.environ.get("OPERATORX_WARMUP_MIN_S", "0.025")) -# The first op of a process starts from an idle GPU; give it a longer ramp. -_FIRST_WARMUP_S = float(os.environ.get("OPERATORX_FIRST_WARMUP_S", "1.0")) -_FIRST = True -# GPU spin enqueued ahead of the timed loop so the CPU queues every timed -# iteration before the first one runs; event brackets then exclude host -# launch overhead. -_SHIELD_CYCLES = int(os.environ.get("OPERATORX_SHIELD_CYCLES", "4000000")) -# Idle time between ops as a multiple of the GPU-busy time, capped per op; -# keeps the duty cycle low enough that every op starts at boost clocks. _COOLDOWN_RATIO = float(os.environ.get("OPERATORX_COOLDOWN_RATIO", "4")) _COOLDOWN_MAX_S = float(os.environ.get("OPERATORX_COOLDOWN_MAX_S", "1.0")) @@ -54,40 +40,6 @@ def _clear_l2() -> None: _L2_BUF[dev].zero_() -def _time_op(fn, sleep_s: float) -> float: - """Median of _ITERS cold, event-timed iterations in us. - - sleep_s > 0 spaces iterations with a host sleep (throttle retry), which - forces a sync per iteration, so each gets its own smaller shield.""" - for _ in range(_WARMUP): - fn() - torch.cuda.synchronize() - global _FIRST - floor, _FIRST = (max(_WARMUP_MIN_S, _FIRST_WARMUP_S) if _FIRST else _WARMUP_MIN_S), False - t0 = time.perf_counter() - while time.perf_counter() - t0 < floor: - fn() - torch.cuda.synchronize() - - starts = [torch.cuda.Event(enable_timing=True) for _ in range(_ITERS)] - ends = [torch.cuda.Event(enable_timing=True) for _ in range(_ITERS)] - if sleep_s <= 0.0: - torch.cuda._sleep(_SHIELD_CYCLES) - for start, end in zip(starts, ends): - if sleep_s > 0.0: - time.sleep(sleep_s) - torch.cuda._sleep(max(_SHIELD_CYCLES // 4, 500000)) - _clear_l2() - start.record() - fn() - end.record() - if sleep_s > 0.0: - torch.cuda.synchronize() - torch.cuda.synchronize() - times = sorted(s.elapsed_time(e) * 1000.0 for s, e in zip(starts, ends)) - return times[_ITERS // 2] - - def run(op: Op) -> Result: _load() impl = _DISPATCH.get((op.type, op.backend)) @@ -96,10 +48,10 @@ def run(op: Op) -> Result: ctx = impl.prepare(op) fn, cuda_graph = impl.launcher(ctx) if impl.launcher else ((lambda: impl.kernel(ctx)), False) - median_us, telem = telemetry.measure(op, lambda sleep_s: _time_op(fn, sleep_s)) + median_us, telem = telemetry.measure(op, lambda sleep_s: timing.time_op(fn, _clear_l2, sleep_s)) if _COOLDOWN_RATIO > 0.0: - busy_s = median_us * 1e-6 * (_ITERS + _WARMUP) + busy_s = median_us * 1e-6 * (timing.ITERS + timing.WARMUP) time.sleep(min(busy_s * _COOLDOWN_RATIO, _COOLDOWN_MAX_S)) metrics = {"latency_us": median_us, "cuda_graph": cuda_graph, "telemetry": telem} @@ -108,4 +60,9 @@ def run(op: Op) -> Result: prof = profiling.profile_op(fn) if prof is not None: metrics["profile"] = prof + per_rank = ranks.summary(metrics) # every rank's timeline and telemetry, on rank 0 + if per_rank is not None: + metrics["ranks"] = per_rank + if per_rank["capped_ranks"] and metrics.get("telemetry"): + metrics["telemetry"]["capped"] = True # any rank capped caps the op return Result(op=op, metrics=metrics) diff --git a/operatorx/runners/nvidia/runtime.py b/operatorx/runners/nvidia/runtime.py index b3acc7e68f..d1ce815892 100644 --- a/operatorx/runners/nvidia/runtime.py +++ b/operatorx/runners/nvidia/runtime.py @@ -1,6 +1,4 @@ -"""NVIDIA-specific runtime probes. ``collect()`` returns software/driver -versions to merge into RunInfo.software. Empty dict when not on NVIDIA or -when probes fail.""" +"""NVIDIA driver version for RunInfo.software; empty when the probe fails.""" from __future__ import annotations diff --git a/operatorx/runtime.py b/operatorx/runtime.py index 9eb5063d14..e33c682721 100644 --- a/operatorx/runtime.py +++ b/operatorx/runtime.py @@ -1,13 +1,6 @@ -"""Capture a snapshot of the runtime environment for one smoke-test invocation. - -Generic host info (hostname, importlib.metadata library versions, git sha, -whitelisted env, run id) lives here. Platform-specific driver/runtime probes -live alongside their kernel runners at ``operatorx/runners//runtime.py`` -and expose a ``collect() -> dict[str, str]`` function whose output is merged -into RunInfo.software. Each probe is fail-soft. - -Env-passed (no detection): cluster, container_image, instance_type via -``$OPERATORX_CLUSTER`` / ``$OPERATORX_CONTAINER_IMAGE`` / ``$OPERATORX_INSTANCE_TYPE``. +"""RunInfo snapshot of the runtime: host info here, driver probes in runners//runtime.py +(collect(), fail-soft). cluster/container_image/instance_type come from $OPERATORX_CLUSTER, +$OPERATORX_CONTAINER_IMAGE, $OPERATORX_INSTANCE_TYPE. """ from __future__ import annotations @@ -65,15 +58,12 @@ def _operatorx_version() -> str: def _generic_software() -> dict[str, str]: - # Host-level only: python version + platform driver/runtime probes. - # Per-backend library versions are merged in by main.py via the backend's - # own ``versions()`` function. + # backend library versions are added by main.py return {"python": sys.version.split()[0]} def _platform_software() -> dict[str, str]: - # Each subpackage of operatorx.runners is a platform. The convention is - # that it ships a `runtime.collect() -> dict[str, str]`. + # each operatorx.runners subpackage is a platform with runtime.collect() out: dict[str, str] = {} for info in pkgutil.iter_modules(_runners_pkg.__path__): if not info.ispkg: diff --git a/operatorx/scripts/consolidate_results.py b/operatorx/scripts/consolidate_results.py index e5df032e39..565f72ac4f 100644 --- a/operatorx/scripts/consolidate_results.py +++ b/operatorx/scripts/consolidate_results.py @@ -1,15 +1,7 @@ #!/usr/bin/env python3 -"""Per-chip dedup pass. - -For each results/// directory: - 1. Read every *.json run file. - 2. For each unique (op_type, args, backend) tuple, keep the row from the - latest run (by run.started_at). - 3. Emit a single new run file with id/timestamp=NOW, run metadata copied - from the most-recent input run for that chip. - 4. Delete the original files. - -Run from the project root (the directory containing `results/`). +"""Per-chip dedup of results///*.json: keep the latest row (by run.started_at) +per (op_type, args, backend) in one new run file, then delete the originals. +Run from the directory containing results/. """ from __future__ import annotations @@ -24,7 +16,6 @@ def _row_key(row: dict) -> tuple: op = row["op"] args = op.get("args", {}) - # Deterministic ordering for dict; args are JSON-loaded so plain dicts. args_t = tuple(sorted(args.items(), key=lambda kv: kv[0])) return (op["type"], op.get("backend"), args_t) @@ -45,7 +36,7 @@ def consolidate_chip(chip_dir: Path, dry_run: bool) -> None: if not files: return - # Map key -> (started_at, row); keep latest. + # key -> (started_at, row) latest: dict[tuple, tuple[str, dict]] = {} latest_run_meta: dict | None = None latest_started = "" @@ -66,7 +57,6 @@ def consolidate_chip(chip_dir: Path, dry_run: bool) -> None: if not latest_run_meta: return - # Build the consolidated run record. cluster = latest_run_meta.get("cluster") new_id = _run_id(cluster) now_iso = _utc_now_iso() @@ -87,7 +77,6 @@ def consolidate_chip(chip_dir: Path, dry_run: bool) -> None: return out_path.write_text(json.dumps(out, indent=2)) - # Remove the originals (excluding the freshly-written one). for f in files: if f != out_path: f.unlink() diff --git a/operatorx/scripts/pull_containers.py b/operatorx/scripts/pull_containers.py deleted file mode 100755 index 8442eb8088..0000000000 --- a/operatorx/scripts/pull_containers.py +++ /dev/null @@ -1,98 +0,0 @@ -#!/usr/bin/env python3 -"""Import container images from containers.toml as enroot squash files. - -Run on the cluster head node before submit_smoke.py. Idempotent — already-imported -images are skipped. The actual import runs inside srun on a compute node, matching -the cluster convention. -""" - -from __future__ import annotations - -import shlex -import subprocess -import sys -from pathlib import Path - -try: - import tomllib # Python 3.11+ -except ModuleNotFoundError: # Python 3.10 fallback (e.g. b300 login) - import tomli as tomllib # type: ignore - -import os - -SQUASH_DIR = Path(os.environ.get("OPERATORX_SQUASH_DIR", "/home/sa-shared/containers")) -PROJECT_ROOT = Path(__file__).resolve().parent.parent -MANIFEST = PROJECT_ROOT / "containers.toml" -PARTITION = os.environ.get("OPERATORX_PARTITION", "gpu-2") -ACCOUNT = os.environ.get("OPERATORX_ACCOUNT") # None -> omit --account -QOS = os.environ.get("OPERATORX_QOS") # None -> omit --qos - - -def safe_name(image: str) -> str: - out = image - for ch in "/:@#": - out = out.replace(ch, "_") - return out - - -def squash_path(image: str) -> Path: - return SQUASH_DIR / f"{safe_name(image)}.sqsh" - - -def unique_images(platform: str) -> set[str]: - data = tomllib.loads(MANIFEST.read_text()) - return {entry["image"] for entry in data.get(platform, {}).values()} - - -def import_image(image: str) -> int: - sqsh = squash_path(image) - if sqsh.exists(): - print(f"[skip] {image} -> {sqsh}") - return 0 - SQUASH_DIR.mkdir(parents=True, exist_ok=True) - - sqsh_q = shlex.quote(str(sqsh)) - lock_q = shlex.quote(f"{sqsh}.lock") - image_q = shlex.quote(image) - inner = ( - f"exec 9>{lock_q} && " - f"flock -w 600 9 || {{ echo 'lock timeout'; exit 1; }} && " - f"if unsquashfs -l {sqsh_q} >/dev/null 2>&1; then " - f"echo '[skip] became valid during lock wait'; " - f"else rm -f {sqsh_q} && enroot import -o {sqsh_q} docker://{image_q}; fi" - ) - - cmd = ["srun", f"--partition={PARTITION}", "--gres=gpu:1", "--time=60", - f"--job-name=pull-{safe_name(image)[:40]}"] - if ACCOUNT: - cmd.append(f"--account={ACCOUNT}") - if QOS: - cmd.append(f"--qos={QOS}") - cmd += ["bash", "-c", inner] - print(f"[pull] {image} -> {sqsh}") - return subprocess.run(cmd).returncode - - -def main(argv: list[str]) -> int: - if len(argv) < 2: - print(f"usage: {argv[0]} ") - print(" e.g., nvidia | amd") - return 2 - platform = argv[1] - images = unique_images(platform) - if not images: - print(f"no entries for platform={platform!r} in {MANIFEST}") - return 1 - print(f"{len(images)} unique image(s) to ensure for {platform}:") - for img in sorted(images): - print(f" {img}") - print() - rc = 0 - for image in sorted(images): - if import_image(image) != 0: - rc = 1 - return rc - - -if __name__ == "__main__": - sys.exit(main(sys.argv)) diff --git a/operatorx/scripts/submit_run.py b/operatorx/scripts/submit_run.py deleted file mode 100755 index 98c88728df..0000000000 --- a/operatorx/scripts/submit_run.py +++ /dev/null @@ -1,192 +0,0 @@ -#!/usr/bin/env python3 -"""Submit SLURM jobs to run operatorx (python -m operatorx) for every backend on a platform. - -For each (image, world_size) combination, submits one srun job that launches -world_size tasks (potentially across multiple nodes). Backends sharing an image -run in the same job; world sizes are launched separately so torch.distributed -can be sized per group. - -Squash files expected at /home/sa-shared/containers/.sqsh. -Run scripts/pull_containers.py first if any are missing. -""" - -from __future__ import annotations - -import json -import shlex -import subprocess -import sys -from collections import defaultdict -from pathlib import Path - -try: - import tomllib # Python 3.11+ -except ModuleNotFoundError: # Python 3.10 fallback (e.g. b300 login) - import tomli as tomllib # type: ignore - -import os - -SQUASH_DIR = Path(os.environ.get("OPERATORX_SQUASH_DIR", "/home/sa-shared/containers")) -PROJECT_ROOT = Path(__file__).resolve().parent.parent -MANIFEST = PROJECT_ROOT / "containers.toml" -DEFAULT_PLATFORM = "nvidia" -PARTITION = os.environ.get("OPERATORX_PARTITION", "gpu-2") -JOB_NAME = os.environ.get("OPERATORX_JOB_NAME", "benchmark") -ACCOUNT = os.environ.get("OPERATORX_ACCOUNT") # None -> omit --account -QOS = os.environ.get("OPERATORX_QOS") # None -> omit --qos -GPUS_PER_NODE = 8 -WORLD_SIZES = [1, 2, 4, 8] # ws>8 disabled: multi-node NCCL IB bring-up hangs on b200/b300 - -# Default cluster ids per platform (overridable via $OPERATORX_CLUSTER). -DEFAULT_CLUSTER = { - "nvidia": "b200_dgx_8x", - "amd": "mi355x_8x", -} - - -def safe_name(image: str) -> str: - out = image - for ch in "/:@#": - out = out.replace(ch, "_") - return out - - -def load_groups(platform: str) -> dict[str, list[str]]: - """Group backends in containers.toml by their container image. - - If $OPERATORX_BACKENDS is set (comma-separated allowlist), only backends - in that list are included; groups that end up empty are dropped. main.py - already honors this var at run time — propagating it here also stops us - from submitting jobs that would have produced zero new rows. - """ - allow = {b.strip() for b in os.environ.get("OPERATORX_BACKENDS", "").split(",") if b.strip()} - data = tomllib.loads(MANIFEST.read_text()) - groups: dict[str, list[str]] = defaultdict(list) - for backend, entry in data.get(platform, {}).items(): - if allow and backend not in allow: - continue - groups[entry["image"]].append(backend) - return dict(groups) - - -def slurm_layout(world_size: int) -> tuple[int, int, int]: - """(nodes, ntasks_per_node, gpus_per_node)""" - if world_size <= GPUS_PER_NODE: - return 1, world_size, world_size - nodes = (world_size + GPUS_PER_NODE - 1) // GPUS_PER_NODE - return nodes, GPUS_PER_NODE, GPUS_PER_NODE - - -def _csv(s: str | None) -> list[str]: - if not s: - return [] - return [x.strip() for x in s.split(",") if x.strip()] - - -def _scan_testlists(testlist_names: list[str]) -> set[int]: - """The world sizes the selected testlists' shapes ask for.""" - tl_dir = PROJECT_ROOT / "testlists" - available = {p.stem: p for p in tl_dir.glob("*.json")} - wanted = testlist_names or sorted(available) - world_sizes: set[int] = set() - for name in wanted: - path = available.get(name) - if path is None: - continue - for shape in json.loads(path.read_text()): - world_sizes.add(int(shape.get("args", {}).get("world_size", 1))) - return world_sizes - - -def submit(image: str, backends: list[str], world_size: int, platform: str, cluster: str) -> int: - sqsh = SQUASH_DIR / f"{safe_name(image)}.sqsh" - if not sqsh.exists(): - print(f"[error] missing squash file: {sqsh}") - print(f" run scripts/pull_containers.py {platform} first") - return 1 - - project = PROJECT_ROOT - nodes, tpn, gpn = slurm_layout(world_size) - job_name = JOB_NAME - backend_list = ",".join(backends) - - inner = " && ".join([ - "set -e", - "export TRITON_CACHE_DIR=/tmp/triton_cache", - "export MPLCONFIGDIR=/tmp/matplotlib_config", - "export HOME=/tmp/home_tmp", - "export NCCL_DEBUG=WARN", - "unset NCCL_ASYNC_ERROR_HANDLING", - "export PYTHONWARNINGS=ignore::SyntaxWarning", - f"cd {project}", - f"export PYTHONPATH={shlex.quote(str(project.parent))}", - "export RANK=$SLURM_PROCID", - "export LOCAL_RANK=$SLURM_LOCALID", - "export WORLD_SIZE=$SLURM_NTASKS", - "export MASTER_ADDR=$(python -c 'import os,re; nl=os.environ.get(\"SLURM_NODELIST\",\"\"); m=re.match(r\"^(.+?)\\[(\\d+)\", nl); print((m.group(1)+m.group(2)) if m else nl)')", - "export MASTER_PORT=29500", - f"OPERATORX_BACKENDS={backend_list} OPERATORX_CLUSTER={cluster}" - + (f" OPERATORX_TESTLISTS={os.environ['OPERATORX_TESTLISTS']}" if os.environ.get("OPERATORX_TESTLISTS") else "") - + " python -m operatorx", - ]) - - # Use sbatch (async) so SLURM can schedule jobs in parallel as resources free. - # `srun --container-image=` inside the wrap lands the actual work in the container. - inner_srun = ( - f"srun --container-image={sqsh} --container-mounts={project}:{project} " - f"bash -c " + shlex.quote(inner) - ) - log_dir = project / "logs" - log_dir.mkdir(parents=True, exist_ok=True) - tag = f"ws{world_size}" - out = log_dir / f"opx-{job_name}-{platform}-{tag}-{'_'.join(backends)}-%j.out" - err = log_dir / f"opx-{job_name}-{platform}-{tag}-{'_'.join(backends)}-%j.err" - cmd = ["sbatch", - f"--partition={PARTITION}", - f"--nodes={nodes}", - f"--ntasks={world_size}", - f"--ntasks-per-node={tpn}", - f"--gres=gpu:{gpn}", - f"--time={os.environ.get('OPERATORX_TIME_MIN', '30')}", - f"--job-name={job_name}", - f"--output={out}", - f"--error={err}"] - if ACCOUNT: - cmd.append(f"--account={ACCOUNT}") - if QOS: - cmd.append(f"--qos={QOS}") - cmd += [f"--wrap={inner_srun}"] - - print(f"[submit] {job_name}") - print(f" image: {image}") - print(f" backends: {backends}") - print(f" world_size: {world_size} ({nodes} node(s) x {tpn} task(s))") - return subprocess.run(cmd).returncode - - -def main(argv: list[str]) -> int: - platform = argv[1] if len(argv) > 1 else DEFAULT_PLATFORM - cluster = os.environ.get("OPERATORX_CLUSTER", DEFAULT_CLUSTER.get(platform, "")) - if not cluster: - print(f"no default cluster for platform={platform!r}; set $OPERATORX_CLUSTER") - return 1 - groups = load_groups(platform) - if not groups: - print(f"no entries for platform={platform!r} in {MANIFEST}") - return 1 - testlist_names = _csv(os.environ.get("OPERATORX_TESTLISTS")) - world_sizes = {ws for ws in _scan_testlists(testlist_names) if ws in WORLD_SIZES} - print(f"{platform}: {len(groups)} container(s) on cluster={cluster}; " - f"{len(world_sizes)} world size(s) -> {len(groups) * len(world_sizes)} job(s)") - print() - rc = 0 - for image, backends in sorted(groups.items()): - for ws in sorted(world_sizes): - if submit(image, sorted(backends), ws, platform, cluster) != 0: - rc = 1 - print() - return rc - - -if __name__ == "__main__": - sys.exit(main(sys.argv)) diff --git a/operatorx/testlists/attn_parallel.json b/operatorx/testlists/attn_parallel.json new file mode 100644 index 0000000000..03edbd7c57 --- /dev/null +++ b/operatorx/testlists/attn_parallel.json @@ -0,0 +1,11884 @@ +[ + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "mxfp4", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "mxfp4", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "mxfp4", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "mxfp4", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "mxfp4", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "mxfp4", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "mxfp4", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "mxfp4", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 0, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 0, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 0, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 0, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 0, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 0, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 0, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 0, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 2, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "indexer": "own", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "reuse", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "shared", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "shared", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "shared", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "shared", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "shared", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "shared", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "shared", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 1, + "source": false, + "indexer": "shared", + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "topk": 512, + "index_heads": 32, + "index_dim": 128, + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "mla", + "args": { + "hidden": 7168, + "heads": 96, + "q_lora_rank": 1536, + "kv_lora_rank": 512, + "nope": 128, + "rope_dim": 64, + "v": 128, + "rope": false, + "gate": true, + "kv_cache_dtype": "fp8", + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn" + ] + }, + { + "type": "mla", + "args": { + "hidden": 7168, + "heads": 96, + "q_lora_rank": 1536, + "kv_lora_rank": 512, + "nope": 128, + "rope_dim": 64, + "v": 128, + "rope": false, + "gate": true, + "kv_cache_dtype": "fp8", + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8, + "dcp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn" + ] + }, + { + "type": "mla", + "args": { + "hidden": 7168, + "heads": 96, + "q_lora_rank": 1536, + "kv_lora_rank": 512, + "nope": 128, + "rope_dim": 64, + "v": 128, + "rope": false, + "gate": true, + "kv_cache_dtype": "fp8", + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn" + ] + }, + { + "type": "mla", + "args": { + "hidden": 7168, + "heads": 96, + "q_lora_rank": 1536, + "kv_lora_rank": 512, + "nope": 128, + "rope_dim": 64, + "v": 128, + "rope": false, + "gate": true, + "kv_cache_dtype": "fp8", + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8, + "dcp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn" + ] + }, + { + "type": "kda", + "args": { + "hidden": 7168, + "heads": 96, + "head_dim": 128, + "conv_kernel": 4, + "state_dtype": "fp32", + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn" + ] + }, + { + "type": "kda", + "args": { + "hidden": 7168, + "heads": 96, + "head_dim": 128, + "conv_kernel": 4, + "state_dtype": "fp32", + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn" + ] + }, + { + "type": "kda", + "args": { + "hidden": 7168, + "heads": 96, + "head_dim": 128, + "conv_kernel": 4, + "state_dtype": "bf16", + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn" + ] + }, + { + "type": "kda", + "args": { + "hidden": 7168, + "heads": 96, + "head_dim": 128, + "conv_kernel": 4, + "state_dtype": "bf16", + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn" + ] + }, + { + "type": "gqa", + "args": { + "hidden": 6144, + "q_heads": 64, + "kv_heads": 4, + "head_dim": 128, + "rope_dim": 64, + "rope_theta": 5000000, + "qk_norm": true, + "kv_cache_dtype": "fp8", + "proj": { + "k_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "o_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "q_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "v_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/self_attn" + ] + }, + { + "type": "gqa", + "args": { + "hidden": 6144, + "q_heads": 64, + "kv_heads": 4, + "head_dim": 128, + "rope_dim": 64, + "rope_theta": 5000000, + "qk_norm": true, + "kv_cache_dtype": "fp8", + "proj": { + "k_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "o_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "q_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "v_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/self_attn" + ] + }, + { + "type": "gqa", + "args": { + "hidden": 6144, + "q_heads": 64, + "kv_heads": 4, + "head_dim": 128, + "rope_dim": 64, + "rope_theta": 5000000, + "qk_norm": true, + "kv_cache_dtype": "fp8", + "proj": { + "k_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "o_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "q_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "v_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/self_attn" + ] + }, + { + "type": "gqa", + "args": { + "hidden": 6144, + "q_heads": 64, + "kv_heads": 4, + "head_dim": 128, + "rope_dim": 64, + "rope_theta": 5000000, + "qk_norm": true, + "kv_cache_dtype": "fp8", + "proj": { + "k_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "o_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "q_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "v_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/self_attn" + ] + }, + { + "type": "gqa", + "args": { + "hidden": 6144, + "q_heads": 64, + "kv_heads": 4, + "head_dim": 128, + "rope_dim": 64, + "rope_theta": 5000000, + "qk_norm": true, + "kv_cache_dtype": "fp8", + "proj": { + "k_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "o_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "q_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "v_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/self_attn" + ] + }, + { + "type": "gqa", + "args": { + "hidden": 6144, + "q_heads": 64, + "kv_heads": 4, + "head_dim": 128, + "rope_dim": 64, + "rope_theta": 5000000, + "qk_norm": true, + "kv_cache_dtype": "fp8", + "proj": { + "k_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "o_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "q_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "v_proj": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 1, + "q": 4096, + "ctx": 0 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/self_attn" + ] + } +] \ No newline at end of file diff --git a/operatorx/testlists/gemm_parallel.json b/operatorx/testlists/gemm_parallel.json new file mode 100644 index 0000000000..21bf5797e0 --- /dev/null +++ b/operatorx/testlists/gemm_parallel.json @@ -0,0 +1,1692 @@ +[ + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 12288, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 12288, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 12288, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 12288, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 12288, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 12288, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/self_attn.o_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/self_attn.o_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/self_attn.o_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/self_attn.o_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/self_attn.o_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/self_attn.o_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "amd/MiniMax-M3-MXFP4/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "amd/MiniMax-M3-MXFP4/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 12288, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "amd/MiniMax-M3-MXFP4/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 12288, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "amd/MiniMax-M3-MXFP4/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 8192, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "amd/MiniMax-M3-MXFP4/self_attn.o_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 8192, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "amd/MiniMax-M3-MXFP4/self_attn.o_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "k": 16384, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro-0813/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "k": 16384, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "k": 16384, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro-0813/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "k": 16384, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "k": 3072, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro-0813/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "k": 3072, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "k": 3072, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro-0813/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "k": 3072, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 8192, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn.wo_b" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 2304, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 2304, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 2304, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 2304, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 2304, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "k": 2304, + "n": 5120, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn.shared_experts.w2" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 6144, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 6144, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 33792, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 33792, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/mlp.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 12288, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn.o_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 256, + "a": { + "dtype": "bf16" + }, + "b": { + "dtype": "bf16" + }, + "k": 12288, + "n": 7168, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn.o_proj" + ] + } +] \ No newline at end of file diff --git a/operatorx/testlists/moe_parallel.json b/operatorx/testlists/moe_parallel.json new file mode 100644 index 0000000000..85fdf0da6d --- /dev/null +++ b/operatorx/testlists/moe_parallel.json @@ -0,0 +1,5680 @@ +[ + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "amd/MiniMax-M3-MXFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "amd/MiniMax-M3-MXFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 4, + "ep": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro-0813/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 4, + "ep": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro-0813/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 4, + "ep": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro-0813/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 4, + "ep": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": false, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "hash", + "vocab": 129280 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro-0813/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 2, + "ep": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4, + "ep": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 2, + "ep": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4, + "ep": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "alpha": 4.0, + "beta": 25.0, + "kind": "situ" + }, + "experts": { + "a1": { + "dtype": "bf16" + }, + "a2": { + "dtype": "bf16" + }, + "inter": 3072, + "latent": 3584, + "latent_norm": true, + "num": 896, + "top_k": 16, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.0, + "scoring": "sigmoid", + "select": { + "groups": 1, + "kind": "grouped_topk", + "topk_groups": 1 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "bf16" + }, + "count": 2, + "inter": 3072, + "w1": { + "dtype": "bf16" + }, + "w2": { + "dtype": "bf16" + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 256, + "activation": { + "alpha": 4.0, + "beta": 25.0, + "kind": "situ" + }, + "experts": { + "a1": { + "dtype": "bf16" + }, + "a2": { + "dtype": "bf16" + }, + "inter": 3072, + "latent": 3584, + "latent_norm": true, + "num": 896, + "top_k": 16, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.0, + "scoring": "sigmoid", + "select": { + "groups": 1, + "kind": "grouped_topk", + "topk_groups": 1 + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "bf16" + }, + "count": 2, + "inter": 3072, + "w1": { + "dtype": "bf16" + }, + "w2": { + "dtype": "bf16" + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/block_sparse_moe" + ] + } +] \ No newline at end of file diff --git a/operatorx/tests/test_ci.py b/operatorx/tests/test_ci.py index 7b4ace0f98..d2a043998b 100644 --- a/operatorx/tests/test_ci.py +++ b/operatorx/tests/test_ci.py @@ -23,8 +23,8 @@ def platforms(runner="cluster:h100-dgxc", gpus=8, architecture="linux/amd64"): def test_plan_chunks_by_world_size(): ordinary = {"type": "gemm", "args": {"m": 2}} - wide = {"type": "gemm", "args": {"m": 2, "world_size": 4}} - multi = {"type": "gemm", "args": {"m": 2, "world_size": 16}} + wide = {"type": "gemm", "args": {"m": 2, "parallel": {"tp": 4}}} + multi = {"type": "gemm", "args": {"m": 2, "parallel": {"tp": 16}}} result = ci.plan( "cluster:h100-dgxc", ["a", "b"], @@ -72,7 +72,7 @@ def test_plan_bounds_world_size_to_physical_arm_node(): shapes = { "tiny": [ {"type": "gemm", "args": {"m": 2}}, - {"type": "allreduce", "args": {"world_size": 8}}, + {"type": "allreduce", "args": {"parallel": {"tp": 8}}}, ] } hardware = platforms("cluster:gb200-nv", 4, "linux/arm64") @@ -115,7 +115,7 @@ def external(argv, **kwargs): monkeypatch.setattr(ci.subprocess, "run", external) digest = "sha256:" + "a" * 64 - # Only registry traffic is replaced; use the production reference parser. + # only registry traffic is faked; the real reference parser runs probe = ci.probe_module() monkeypatch.setattr(probe, "resolve_image_digest", lambda image: digest) monkeypatch.setattr(ci, "probe_module", lambda: probe) @@ -137,7 +137,7 @@ def external(argv, **kwargs): args.image_platform = architecture monkeypatch.setattr(ci.platform, "machine", lambda: machine) ci.import_image(args) - ci.import_image(args) # Reuse a validated cache on the same architecture. + ci.import_image(args) # reuses the validated cache images = list(args.cache.glob("*.sqsh")) assert len(images) == 2 assert all(image.read_bytes() == b"validated squash fixture" for image in images) @@ -201,7 +201,7 @@ def test_platform_overlay_preserves_base_and_replaces_explicit_profile(tmp_path) [("flashinfer", [1], "gemm"), ("torch", [2], "gemm"), ("torch", [1], "allreduce")], ) def test_amd_plan_rejects_unimplemented_execution(backend, worlds, kind): - with pytest.raises(ValueError, match="single-GPU torch/vllm GEMM"): + with pytest.raises(ValueError, match="AMD CI supports|single-device"): ci.plan( "cluster:mi300x-amd", [backend], @@ -220,8 +220,7 @@ def test_amd_plan_rejects_unimplemented_execution(backend, worlds, kind): def test_strict_benchmark_writes_actual_status( tmp_path, monkeypatch, outcome, expected_rc ): - # The GPU kernel is an external collaborator; selection, exception handling, - # checkpointing, serialization and exit decisions execute the real main(). + # only the GPU kernel is faked; everything else is the real main() backend = types.ModuleType("operatorx.runners.testgpu.backends.kernel") backend.IMPLS = ( [] if outcome == "unclaimed" else [types.SimpleNamespace(op_type="gemm")] From 0c27562ad2f2f10822a7f8f2a4fe2ee8b083370e Mon Sep 17 00:00:00 2001 From: hbarclay Date: Mon, 28 Sep 2026 00:15:28 -0400 Subject: [PATCH 2/6] operatorx engine: the recipe's CUDA-graph cap moves into its compilation config vLLM rejects max_cudagraph_capture_size given both as its own argument and in the compilation config, which a non-eager engine now receives. Co-Authored-By: Claude Opus 5.5 (1M context) --- operatorx/runners/common/vllm/engine.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/operatorx/runners/common/vllm/engine.py b/operatorx/runners/common/vllm/engine.py index 7c5209067f..2c305c039d 100644 --- a/operatorx/runners/common/vllm/engine.py +++ b/operatorx/runners/common/vllm/engine.py @@ -49,8 +49,8 @@ def recipe_kwargs(without: tuple = ()) -> tuple[dict, dict]: if k not in without} attention = json.loads(args.pop("attention-config", None) or "{}") compilation = json.loads(args.pop("compilation-config", None) or "{}") - if args.get("max-cudagraph-capture-size"): - compilation.setdefault("max_cudagraph_capture_size", int(args["max-cudagraph-capture-size"])) + if args.get("max-cudagraph-capture-size"): # moved: vLLM rejects it in both places + compilation.setdefault("max_cudagraph_capture_size", int(args.pop("max-cudagraph-capture-size"))) info = {"attention_config": attention, "compilation_config": compilation} if not args: return {}, info From 02ebb31624fdcd938257c609c7d46092bbe16e3d Mon Sep 17 00:00:00 2001 From: hbarclay Date: Mon, 28 Sep 2026 00:15:47 -0400 Subject: [PATCH 3/6] operatorx: smoke_parallel, one case per op family and parallel split Co-Authored-By: Claude Opus 5.5 (1M context) --- operatorx/testlists/smoke_parallel.json | 2083 +++++++++++++++++++++++ 1 file changed, 2083 insertions(+) create mode 100644 operatorx/testlists/smoke_parallel.json diff --git a/operatorx/testlists/smoke_parallel.json b/operatorx/testlists/smoke_parallel.json new file mode 100644 index 0000000000..9487e31364 --- /dev/null +++ b/operatorx/testlists/smoke_parallel.json @@ -0,0 +1,2083 @@ +[ + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "gemm", + "args": { + "m": 8, + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "k": 3072, + "n": 6144, + "out": "bf16", + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe.shared_experts.down_proj" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "MiniMaxAI/MiniMax-M3-MXFP8/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "alpha": 1.702, + "beta": 1.0, + "kind": "swigluoai", + "limit": 7.0 + }, + "experts": { + "a1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "a2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": false + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "inter": 3072, + "num": 128, + "top_k": 4, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "e4m3", + "group": [ + 1, + 16 + ], + "static": true + }, + "scale2": { + "dtype": "fp32", + "group": [ + -1, + -1 + ], + "static": true + } + } + }, + "hidden": 6144, + "router": { + "bias": true, + "gate": { + "dtype": "fp32" + }, + "renormalize": true, + "scale": 2.0, + "scoring": "sigmoid", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "nvidia/MiniMax-M3-NVFP4/block_sparse_moe" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 4, + "ep": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 3072, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 7168, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 2.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "count": 1, + "inter": 3072, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro-0813/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 2, + "ep": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 4, + "ep": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "moe", + "args": { + "tokens": 8, + "activation": { + "kind": "silu", + "limit": 10.0 + }, + "experts": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "a2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "inter": 2304, + "num": 384, + "top_k": 6, + "w1": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e2m1", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": true + } + } + }, + "hidden": 5120, + "router": { + "bias": true, + "gate": { + "dtype": "bf16", + "logits": "fp32" + }, + "renormalize": true, + "scale": 1.5, + "scoring": "sqrtsoftplus", + "select": { + "kind": "topk" + } + }, + "routing": { + "distribution": "natural", + "seed": 0 + }, + "shared": { + "a1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "count": 1, + "inter": 2304, + "w1": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + }, + "w2": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "parallel": { + "tp": 8, + "ep": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/ffn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 4 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 7168, + "heads": 128, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1536, + "window": 128, + "o_groups": 16, + "o_lora_rank": 1024, + "compress_rope_theta": 160000, + "rope_scaling": { + "type": "yarn", + "factor": 16, + "original_max": 65536, + "beta_fast": 32, + "beta_slow": 1 + }, + "kv_cache_dtype": "fp8", + "compress_ratio": 4, + "indexer": "own", + "topk": 1024, + "index_heads": 64, + "index_dim": 128, + "index_cache_dtype": "fp8", + "proj": { + "indexer.wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 128 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 128, + 128 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "dp": 8 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4-Pro/attn" + ] + }, + { + "type": "dsv4_attn", + "args": { + "hidden": 5120, + "heads": 64, + "head_dim": 512, + "rope_dim": 64, + "q_lora_rank": 1280, + "window": 128, + "o_groups": 8, + "o_lora_rank": 1024, + "compress_ratio": 0, + "proj": { + "wkv": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wo_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_a": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + }, + "wq_b": { + "a": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 1, + 32 + ], + "static": false + } + }, + "b": { + "dtype": "e4m3", + "scale": { + "dtype": "ue8m0", + "group": [ + 32, + 32 + ], + "static": true + } + } + } + }, + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 2 + } + }, + "sources": [ + "deepseek-ai/DeepSeek-V4.1-Flash/attn" + ] + }, + { + "type": "mla", + "args": { + "hidden": 7168, + "heads": 96, + "q_lora_rank": 1536, + "kv_lora_rank": 512, + "nope": 128, + "rope_dim": 64, + "v": 128, + "rope": false, + "gate": true, + "kv_cache_dtype": "fp8", + "batch": { + "groups": [ + { + "count": 16, + "q": 1, + "ctx": 16384 + } + ] + }, + "parallel": { + "tp": 8, + "dcp": 8 + } + }, + "sources": [ + "moonshotai/Kimi-K3/self_attn" + ] + } +] \ No newline at end of file From 9747d3d4ddb72cfa213eacd3af82a5399ba570e7 Mon Sep 17 00:00:00 2001 From: hbarclay Date: Mon, 28 Sep 2026 01:03:21 -0400 Subject: [PATCH 4/6] operatorx vllm: images older than the pinned vLLM Attention sets enable_jit_warmup only where the image's KernelConfig has it; a MoE case on an image whose vLLM lacks the API the generic block uses is unsupported, not an error (the DeepSeek-V4 DEP recipe image runs vLLM 0.17). Co-Authored-By: Claude Opus 5.5 (1M context) --- operatorx/runners/common/vllm/attention.py | 17 ++++++++++------- operatorx/runners/common/vllm/moe.py | 2 ++ 2 files changed, 12 insertions(+), 7 deletions(-) diff --git a/operatorx/runners/common/vllm/attention.py b/operatorx/runners/common/vllm/attention.py index 700b73e7e4..ef246f94a5 100644 --- a/operatorx/runners/common/vllm/attention.py +++ b/operatorx/runners/common/vllm/attention.py @@ -13,7 +13,7 @@ import random import sys import tempfile -from dataclasses import dataclass, field +from dataclasses import dataclass, field, fields from pathlib import Path import torch @@ -283,12 +283,15 @@ def __init__(self, key: str, b: _Build, split: dict | None): model=self.dir, load_format="dummy", skip_tokenizer_init=True, enforce_eager=True, enable_prefix_caching=False, max_num_seqs=_MAX_SEQS, kv_cache_memory_bytes=_kv_bytes(), gpu_memory_utilization=_GPU_UTIL, **b.engine) - # JIT kernels on first use (the untimed step), not every shape up front - kernel = kwargs.get("kernel_config") - if kernel is None or isinstance(kernel, dict): - kwargs["kernel_config"] = {**(kernel or {}), "enable_jit_warmup": False} - else: - kernel.enable_jit_warmup = False + # JIT kernels on first use (the untimed step), not every shape up front, where the + # image's vLLM has that setting + from vllm.config.kernel import KernelConfig + if "enable_jit_warmup" in {f.name for f in fields(KernelConfig)}: + kernel = kwargs.get("kernel_config") + if kernel is None or isinstance(kernel, dict): + kwargs["kernel_config"] = {**(kernel or {}), "enable_jit_warmup": False} + else: + kernel.enable_jit_warmup = False # the engine runs eagerly: the full graphs serving would capture, from vLLM self.capture = vllm_engine.full_graph_sizes(kwargs, self.recipe["compilation_config"]) self.reqs: list = [] diff --git a/operatorx/runners/common/vllm/moe.py b/operatorx/runners/common/vllm/moe.py index 50dce271db..e134c685bd 100644 --- a/operatorx/runners/common/vllm/moe.py +++ b/operatorx/runners/common/vllm/moe.py @@ -404,6 +404,8 @@ def _prepare_moe(op: Op) -> dict: try: own = modules.build(a) if os.environ.get("OPERATORX_MODEL_MODULES") == "1" else None block = _ModelBlock(own) if own is not None else _generic_block(a, prefix) + except ImportError as e: # an image whose vLLM predates the MoE API this block uses + raise UnsupportedOpError(f"this image's vLLM has no {e.name or e}"[:400]) from e except (NotImplementedError, AssertionError, ValueError, RuntimeError, TypeError, KeyError, AttributeError) as e: if _is_fault(e) or (isinstance(e, AttributeError) and not _missing_op(e)): From 72cc1e86848d7b56b9e323c2f50d62f153ba74f6 Mon Sep 17 00:00:00 2001 From: hbarclay Date: Mon, 28 Sep 2026 01:13:28 -0400 Subject: [PATCH 5/6] operatorx engine: expert parallelism stays the recipe's for ops that don't split along it An attention case on the DeepSeek-V4 DEP recipe (DP attention, MegaMoE experts) kept the recipe's MoE backend but lost its expert parallelism, which MegaMoE requires. Co-Authored-By: Claude Opus 5.5 (1M context) --- operatorx/runners/common/vllm/attention.py | 3 ++- operatorx/runners/common/vllm/engine.py | 9 ++++++--- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/operatorx/runners/common/vllm/attention.py b/operatorx/runners/common/vllm/attention.py index ef246f94a5..d1adca2550 100644 --- a/operatorx/runners/common/vllm/attention.py +++ b/operatorx/runners/common/vllm/attention.py @@ -18,7 +18,7 @@ import torch -from operatorx.core import BackendImpl, Op, UnsupportedOpError +from operatorx.core import BackendImpl, Op, UnsupportedOpError, parallel from operatorx.runners.common import ranks from operatorx.runners.common.vllm import engine as vllm_engine from operatorx.runners.common.vllm import linear as vllm_linear @@ -280,6 +280,7 @@ def __init__(self, key: str, b: _Build, split: dict | None): vllm_engine.launch() kwargs, self.recipe = vllm_engine.engine_args( split, without=b.without, eager=True, defaults={"max_num_batched_tokens": _MAX_BATCHED_TOKENS}, + axes=parallel.ATTENTION_AXES, model=self.dir, load_format="dummy", skip_tokenizer_init=True, enforce_eager=True, enable_prefix_caching=False, max_num_seqs=_MAX_SEQS, kv_cache_memory_bytes=_kv_bytes(), gpu_memory_utilization=_GPU_UTIL, **b.engine) diff --git a/operatorx/runners/common/vllm/engine.py b/operatorx/runners/common/vllm/engine.py index 2c305c039d..5f14fc6029 100644 --- a/operatorx/runners/common/vllm/engine.py +++ b/operatorx/runners/common/vllm/engine.py @@ -64,10 +64,11 @@ def recipe_kwargs(without: tuple = ()) -> tuple[dict, dict]: def engine_args(split: dict | None, without: tuple = (), eager: bool = False, defaults: dict | None = None, - **overrides) -> tuple[dict, dict]: + axes: tuple = parallel.AXES, **overrides) -> tuple[dict, dict]: """EngineArgs kwargs for a case: defaults, the recipe's, the split, the overrides (an attention_config override merges into the recipe's); plus the recipe's attention and - compilation configs.""" + compilation configs. Expert parallelism stays the recipe's when "ep" is not among the + op's axes (it does not change the world size).""" kw, info = recipe_kwargs(without) kw = {**(defaults or {}), **kw} compilation = info["compilation_config"] @@ -76,8 +77,10 @@ def engine_args(split: dict | None, without: tuple = (), eager: bool = False, de elif compilation.get("custom_ops"): # custom-op selection holds without compilation kw["compilation_config"] = {"custom_ops": compilation["custom_ops"]} key = parallel.normalize(split) + recipe = json.loads(os.environ.get("OPERATORX_ENGINE_ARGS") or "{}") + ep = key["ep"] > 1 if "ep" in axes else bool(recipe.get("enable-expert-parallel")) kw.update(tensor_parallel_size=key["tp"], data_parallel_size=key["dp"], - decode_context_parallel_size=key["dcp"], enable_expert_parallel=key["ep"] > 1, + decode_context_parallel_size=key["dcp"], enable_expert_parallel=ep, distributed_executor_backend="external_launcher") attention = {**info["attention_config"], **overrides.pop("attention_config", {})} kw.update(overrides) From 425b44b254a26a92fbc0c427234fa2721743e745 Mon Sep 17 00:00:00 2001 From: hbarclay Date: Mon, 28 Sep 2026 12:50:28 -0400 Subject: [PATCH 6/6] operatorx attention: DeepSeek-V4-Pro's untimed experts keep the checkpoint's width Co-Authored-By: Claude Opus 5.5 (1M context) --- operatorx/runners/common/vllm/attention.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/operatorx/runners/common/vllm/attention.py b/operatorx/runners/common/vllm/attention.py index d1adca2550..d463ce999a 100644 --- a/operatorx/runners/common/vllm/attention.py +++ b/operatorx/runners/common/vllm/attention.py @@ -247,8 +247,8 @@ def _build_dsv4_pro(a: dict) -> _Build: qk_rope_head_dim=a["rope_dim"], q_lora_rank=a["q_lora_rank"], sliding_window=a["window"], o_groups=a["o_groups"], o_lora_rank=a["o_lora_rank"], compress_ratios=ratios, rope_theta=a.get("rope_theta", 10000.0), num_hidden_layers=len(ratios), num_nextn_predict_layers=0, - num_hash_layers=0, n_routed_experts=8, num_experts_per_tok=2, moe_intermediate_size=_MLP, - vocab_size=_VOCAB) + num_hash_layers=0, n_routed_experts=8, num_experts_per_tok=2, vocab_size=_VOCAB) + # the checkpoint's expert width: DeepGEMM's MegaMoE (DEP recipes) rejects a narrower one if a["compress_ratio"]: c["compress_rope_theta"] = a["compress_rope_theta"] if a.get("rope_scaling"):