diff --git a/docs/data-transforms.md b/docs/data-transforms.md index 5fef22b39..f5dcd0935 100644 --- a/docs/data-transforms.md +++ b/docs/data-transforms.md @@ -64,7 +64,26 @@ Returns `{ chartData: InferenceData[][], hardwareConfig: HardwareConfig }`. - `tpPerMw` — `(tputPerGpu * 1000) / hardwarePower` (GPU power in kW, result in tok/s/MW). - Cost-per-million fields — GPU hourly cost divided by tokens-per-hour (in millions): `costh` / `costr` for owning-at-large-hyperscaler-volume / 3-year-rental pricing respectively. Three token variants exist: combined (`costh`/`costr`), output-only (`costhOutput`/`costrOutput`), and input-only (`costhi`/`costri`). The former Neocloud tier (`costn`/`costni`/`costnOutput`) was removed; its share-link keys alias to the hyperscaler-volume metric. - Infrastructure purchasing-power fields divide tokens per hour by the same hourly costs. Total (`tokensPerDollarH`/`R`), output-only (`outputTokensPerDollarH`/`R`), and input-only (`inputTokensPerDollarH`/`R`) are separate USD Y-axis metrics. The former ¥-priced `tokensPerRmb*` axes and the Neocloud `*PerDollarN` axes were removed; their share-link keys alias to the matching hyperscaler-volume $ metric. The dashboard defaults to the hyperscaler-volume total-tokens-per-dollar axis (`y_tokensPerDollarH`). Links created for the removed API-price `i_metric=y_tokensPerDollar` axis resolve to the same hyperscaler-volume variant. -- Energy fields — `jTotal` / `jOutput` / `jInput`: `(hardwarePower * 1000) / tputPerGpu` (Joules per token, where power in kW is converted to W). +- Energy fields — `jTotal` / `jOutput` / `jInput`: `(hardwarePower * 1000) / tputPerGpu` (Joules per token, where power in kW is converted to W). These divide one GPU's all-in power by that GPU's reported throughput; for disaggregated rows `output_tput_per_gpu` is per decode GPU, so `jOutput` ignores the prefill pool. Its semantics are frozen; the whole-deployment view lives in the power boundaries below. +- Power-boundary fields — see [Power boundaries](#power-boundaries). + +#### Power boundaries + +`lib/power-basis.ts` derives the three boundaries beyond GPU-measured telemetry as W per allocated GPU plus J per successful output token. `buildDerivedChartFields` calls `buildPowerBasisChartFields(entry, specs)` right after the measured fields, so the official chart, the `?unofficialrun=` overlay (`transformBenchmarkRows`), and Historical Trends (`rowToLightweightPoint` with `requestedMetrics`) share one formula set; each key is gated by `wants(key)` so selective output stays exact. + +| Basis | W / GPU | J / output token | Fields | +| --------------------------------------------- | -------------------------------------------------------------------------- | ---------------------------------- | ------------------------------------------------------------------ | +| B1 GPU measured | `avg_power_w` | `joules_per_output_token` | existing `measuredAvgPower`, `measuredJPerOutputToken` (unchanged) | +| B2 GPU provisioned | `HW_REGISTRY.tdp` | `W × N_alloc ÷ total output tok/s` | `gpuProvisionedWatts`, `gpuProvisionedJPerOutputToken` | +| B3 utility provisioned (all-in) | `HW_REGISTRY.power × 1000` | `W × N_alloc ÷ total output tok/s` | `utilityProvisionedWatts`, `utilityProvisionedJPerOutputToken` | +| B4 utility modeled (measured → chassis → PUE) | `modeledSystemPower.deploymentFacilityWatts ÷ modeledSystemPower.gpuCount` | `B1 J/out × (B4 W ÷ B1 W)` | `utilityModeledWatts`, `utilityModeledJPerOutputToken` | + +- **N_alloc and total throughput** (`powerBasisNormalization`). Aggregate rows report output per allocated GPU, so `N_alloc` cancels and the builder uses `W ÷ output_tput_per_gpu` without trusting display counts (legacy ingest can encode TP × EP twice). Fixed-sequence disaggregated rows (`disagg && benchmark_type === 'single_turn'`) report output per decode GPU while the deployment also powers the prefill pool, so `total output tok/s = output_tput_per_gpu × num_decode_gpu` and `N_alloc = num_prefill_gpu + num_decode_gpu`; the article's 4P+4D counts eight GPUs, and B3 energy is `(P + D) / D` times `jOutput` on such rows. Other disaggregated benchmark types emit no provisioned energy because it is not verifiable in-app whether AgentX throughput already divides by all GPUs. +- **B4 source.** Reuses the `SystemPowerEstimate` that `rowToAggDataEntry` attached as `entry.modeledSystemPower`; nothing re-runs `modelSystemPower`. `deploymentFacilityWatts` is chassis AC × PUE with PUE applied exactly once inside `estimateChassisPower`, divided by the physical measured GPU count (not `modeledGpuCount`, which over-counts partially allocated chassis: a 4P+4D deployment on two worker hosts models 16 GPUs while measuring 8, and the unit test pins that divisor). This is a different quantity from the existing `modeledChassisPowerPerGpu` (chassis AC ÷ modeled GPU count, no PUE). Modeled energy scales the producer's same-window `joules_per_output_token` by `B4 W ÷ avg_power_w`, so it inherits B1's token denominator and survives rows whose `output_tput_per_gpu` is missing (the provisioned energies do not; a per-basis point count cannot assume one shared denominator). +- **Null rules.** A value is emitted only when finite and positive; otherwise the key is omitted (never `{ y: 0 }`), because the metric filters drop points by `metricKey in point` and `remapInferencePoint` falls back to raw throughput when a key exists with an unusable value. B2/B3 W are absent for hardware without registry specs (`getGpuSpecs` returns zeros); their energies are also absent without output throughput or disaggregated counts. B4 requires `modeledSystemPower.status === 'supported'` and B1 watts (`avg_power_w` after `rowToAggDataEntry`'s `power_valid !== 0` admission); B4 energy additionally needs `joules_per_output_token`. Telemetry admission belongs to `modelSystemPower`, which accepts `power_valid === 1` with schema v2 or the validated unversioned single-node producer (`telemetryBasis: 'validated-unversioned-single-node'`), so B4 renders on exactly the rows that show B1 and `modeledChassisPowerPerGpu`; off-8k/1k workloads, unsupported hardware such as GB200/GB300 NVL72 (`reason: 'hardware'`), and telemetry/topology failures withhold it. The public API's `strictV2` row filter is not re-applied in the chart for B1, so it is not re-applied for B4 either (plan §3.1 describes B1 with that filter; the app's chart path is the authority here). +- **Ordering invariant** (unit-tested on real B200 telemetry): B3 ≥ B4 ≥ B1 and B2 ≥ B1 for both W/GPU and J/out on the same point. +- **Reconstructed prefill energy** (`reconstructedPrefillJPerOutputToken`, `utils/role-energy.ts`). For validated (`power_valid === 1`, schema 2) disaggregated rows, `prefill_joules_per_input_token × (joules_per_output_token ÷ joules_per_input_token)` carries the prefill pool's energy onto the output-token axis: the ratio is the served input:output token count because schema-2 aggregate energy has one numerator. With `decode_joules_per_output_token` it sums back to the deployment's J/out. It feeds only the `i_pcompare=roles` comparison on the energy axis (PowerX Figure 7) and is never a y-axis of its own; aggregate rows and rows missing any of the four inputs omit it. +- Historical Trends substitutes `output_tput_per_gpu := tput_per_gpu` for legacy rows lacking output throughput; provisioned energies in trends inherit that fallback. **GPU specs lookup** happens inside `buildDerivedChartFields`. `getGpuSpecs(hwKey)` (`lib/constants.ts`) splits on `[-_]` to extract the base GPU token (for example, `"b200_trt_mtp"` becomes `"b200"`) and looks it up in `HW_REGISTRY`. Missing keys return zeroed specs, producing `0` cost/energy values rather than crashing. diff --git a/docs/index.md b/docs/index.md index c67e877ae..2e6c0224a 100644 --- a/docs/index.md +++ b/docs/index.md @@ -9,6 +9,7 @@ Design rationale and non-obvious conventions. See [CLAUDE.md](../CLAUDE.md) for - [API Skill Examples](./inferencex-api-examples.md) — Install the public skill, query benchmarks, export measured PowerX data, and explain empty results - [PowerX System Power](./powerx-system-power.md) — Pinned chassis model, measured-input guards, assumptions, and reproducible article exports +- [PowerX Permanent View](./powerx-permanent-view.md) — Power boundaries as gated Measured Energy metrics, `i_metric`/`i_rulers` share links, missing-value states - [PowerX Persistence and Recovery](./powerx-persistence-recovery.md) — Telemetry receipts, migration prerequisites, and targeted repair - [API Skill Releases](./inferencex-skills-release.md) — Prepare an immutable package, verify clean installations and agent exports, and publish through the package-specific workflow - [API Skill Discovery](./inferencex-skills-discovery.md) — Accept or reject implicit skill discovery in fresh Codex and Claude Code projects diff --git a/docs/powerx-permanent-view.md b/docs/powerx-permanent-view.md new file mode 100644 index 000000000..8e3725ca2 --- /dev/null +++ b/docs/powerx-permanent-view.md @@ -0,0 +1,286 @@ +# PowerX Permanent View + +How the PowerX article's power-boundary figures live inside `/inference` instead of a +separate page. The measured-power UI, the four power boundaries, and their share links all +sit in the existing ↑↑↓↓-gated **Measured Energy** metric group. + +## Purpose + +The article compares one deployment's power at four boundaries, from the GPU board to the +utility meter, both as W per GPU and as J per output token. The dashboard already had +GPU-measured telemetry (`measured*` metrics) and an ungated all-in provisioned energy +(`jOutput`). This view adds the other boundaries as ordinary y-axis metrics so every chart +feature (zoom, tooltip, Perf Ruler, Optimal Only, Table, CSV, PNG, `?unofficialrun=` overlay, +share URL) applies unchanged. One chart engine, no second chart component. + +## Gate + +The Measured Energy group is `gated: true` in `metric-registry.ts`. `ChartControls` hides a +gated group while the gate is locked unless the selected `i_metric` belongs to it, so a +shared link to any boundary renders for a reader who never unlocked the gate, while the +group stays out of the selector otherwise. The boundary metrics are members of that group +(`POWER_BASIS_METRIC_CONFIG_KEYS` is spread into it) so they inherit exactly this behaviour. + +## Boundaries + +| Basis (`PowerBasis`) | Selector label | W / GPU metric | J / output token metric | Source | +| --------------------- | ---------------------------- | -------------------------------------------- | ------------------------------------------------- | ----------------------------------------------------------------------- | +| `gpu-measured` | GPU measured | `y_measuredAvgPower` (+P75/P90, roles, %TDP) | `y_measuredJPerOutputToken` (+ input/total/query) | runner telemetry; existing metrics, unchanged | +| `gpu-provisioned` | GPU provisioned (TDP) | `y_gpuProvisionedWatts` | `y_gpuProvisionedJPerOutputToken` | `HW_REGISTRY.tdp` | +| `utility-provisioned` | Utility provisioned (all-in) | `y_utilityProvisionedWatts` | `y_utilityProvisionedJPerOutputToken` | `HW_REGISTRY.power` (all-in kW per GPU) | +| `utility-modeled` | Utility modeled (PUE) | `y_utilityModeledWatts` | `y_utilityModeledJPerOutputToken` | `modelSystemPower` chassis AC × PUE 1.3 (applied once), ÷ measured GPUs | + +Formulas, the all-GPU normalization (`N_alloc` = prefill + decode GPUs for disaggregated +rows) and the null rules are specified in +[Data Transforms → Power boundaries](./data-transforms.md#power-boundaries); the chassis model +itself in [PowerX System Power](./powerx-system-power.md). The ungated `jOutput` keeps its +per-decode-GPU normalization; its labels and the boundary metric's `all GPUs` label keep the +two distinguishable in the selector, the availability list and CSV headers. + +Every boundary metric has `polarity: 'lower'`. Energy metrics therefore get the usual +lower-is-better Pareto frontier. The three watt keys are listed in `POWER_CURVE_METRICS` +(`utils/powerCurves.ts`), so `ScatterGraph`/`GPUGraph` draw the upper power envelope for them +exactly as for `y_measuredAvgPower`, and Optimal Only keeps every load point; the declared +`lower_*` roofline only drives the ascending Table sort. They stay out of +`isMeasuredPowerCurveMetric`, like `y_modeledChassisPowerPerGpu`, so telemetry-only decorations +do not apply. `measured-power-direction.test.ts` pins both halves. A flat provisioned series +(TDP or all-in W is one constant per hardware) has a single envelope vertex per hardware, so no +envelope line is drawn for it — only the points. + +## Why the metric key carries the boundary + +There is no `i_pbasis` URL parameter. The Boundary select in `MeasuredMetricControls` is +presentation over `selectedYAxisMetric`, exactly like Per / Scope / Statistic / Display / +Unit: `measured-metric-config.ts` gives every Measured Energy key a `basis` and resolves each +control change to the nearest registered key. Consequences: + +- One share link per figure, and `i_metric` already round-trips through every existing + surface (share button, Historical Trends hand-off, Table, CSV header). +- Existing links are byte-identical: all `measured*` keys carry `basis: 'gpu-measured'`, and + `MEASURED_METRIC_DEFAULTS` did not change. +- Choosing a derived boundary snaps to its canonical combination (whole deployment, average + watts, joules per output token). Changing any telemetry-only setting while on a derived + boundary returns to GPU measured, so no control ever points at a key that does not exist. +- A gated metric selected through a shared URL keeps its Measured controls, as the + `measured*` siblings already do. +- The Y-axis dropdown's collapsed "Measured Power" / "Measured Energy" option for the other + family resolves to `MEASURED_METRIC_DEFAULTS[family]` (GPU measured), so switching family + from the dropdown resets the boundary; the resolver itself keeps `basis` on a family change + (`changeMeasuredMetricConfig(key, { family })`), which is what a future family control in + the Measured row should call. + +The Boundary select tracks `inference_power_basis_changed { basis, family }` in addition to the +existing `inference_y_axis_metric_selected` fired by `ChartControls`. + +## Missing values + +A point without a value for the selected boundary is omitted from that series only (the +builders never emit `{ y: 0 }`, and both the official and overlay paths filter by +`metricKey in point`). The PowerX availability panel explains the gap per point: + +| State | Meaning | +| ------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `noSpec` | hardware has no `tdp` / `power` in `HW_REGISTRY` | +| `noThroughput` | provisioned watts exist but the row has no output throughput | +| `noNormalization` | disaggregated row with throughput but no whole-deployment GPU count (`powerBasisNormalization`: not `single_turn`, or non-integer prefill/decode counts) | +| `noTelemetry` | modeled boundary needs validated GPU telemetry (B4 follows B1) | +| `invalid` | `power_valid === 0` | +| `modelWorkload` | chassis model covers 8K / 1K only | +| `modelHardware` | hardware outside the chassis profiles (GB200 / GB300 NVL72) | +| `modelUnsupported` | other `modelSystemPower` reasons (gpu-count, topology, role-power …) | + +The chart caption (`data-testid="power-basis-assumptions"`) names the boundary, the formula in +words, the PUE constant and the chassis-model revision so a screenshot records its method. +The pinned tooltip's "Modeled system power" block (`tooltipUtils.ts` `modeledSystemPowerHTML`) +renders only for `measured*` keys and `y_modeledChassisPowerPerGpu`, so on the boundary keys the +caption is the only per-chart provenance; the caption does not promise more. +The empty state for the utility-modeled boundary says why nothing rendered and points at the +Boundary select. + +## Article figure → share link + +Base: `/inference?g_model=&i_seq=8k/1k&i_prec=` plus the hardware legend +selection. Append: + +| Figure | Content | Parameters | +| ------ | --------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| 2 | H200 W/GPU, all four boundaries on one chart | `&i_metric=y_measuredAvgPower&i_pcompare=boundaries` (one boundary at a time: `&i_metric=y_gpuProvisionedWatts` / `y_utilityProvisionedWatts` / `y_utilityModeledWatts`) | +| 3 | H200 J/output token, all four boundaries | `&i_metric=y_measuredJPerOutputToken&i_pcompare=boundaries` (single boundary: the matching `…JPerOutputToken` key) | +| 4 / 5 | B200 vs B300, GB200 vs GB300 at fixed X | `&i_metric=&i_rulers=` (Perf Ruler, T1) | +| 6 | Prefill vs decode W/GPU with the whole deploy | `&i_metric=y_measuredAvgPower&i_pcompare=roles` | +| 7 | Reconstructed request energy, prefill share | `&i_metric=y_measuredJPerOutputToken&i_pcompare=roles` (prefill share in the clone's tooltip) | +| 1 | Measured power over the benchmark job | `&i_metric=y_measuredPowerTimeline` (Display → Timeline, see below) | + +Pinned official dashboard points now offer **View PowerX** when the feature gate is unlocked +or a measured-metric share link is active. The lazy in-page dialog reads the existing +`GET /api/v1/gpu-metrics-point?id=N` API, without entering a run ID or changing chart filters. +This also works in the hardware/date comparison view. The per-chip chart and statistics span +the recorded job (including startup/warmup), not just the audited serving window. Fixed-sequence +points do not request AgentX server-metric overlays. Missing telemetry is shown as unavailable, +not zero power. Unofficial points have no database ID and retain **View power trace**, which +resolves their run/audit provenance automatically. No API contract or ingestion changes are +needed for this presentation-only entry point. + +The `/gpu-metrics` page keeps the raw per-run explorer (every metric, one artifact at a time); +the timeline below is the chart-scoped view of the same artifacts. + +## Comparison series (`i_pcompare`) + +`i_pcompare=boundaries|roles` (default `''`, "Off") overlays sibling series on the selected +metric without changing it. `utils/power-compare.ts` is the single source: `useChartData` and +the `?unofficialrun=` processor (`processOverlayChartDataWithClipping`) both call +`expandPowerCompareSeries`, which appends one clone per sibling to every base point — same +`x`, `y` taken from the sibling's field, `powerVariant` set — and leaves the base points +untouched, so a chart without a comparison is byte-identical to before. A point lacking a +sibling's value contributes nothing to that series (never a 0), so each boundary's and +role's availability rules carry through unchanged. + +| Mode | Base series | Siblings | Fields | +| ------------ | ------------------------------------------------- | -------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `boundaries` | the selected boundary (`basis` of the metric key) | the other three boundaries | `measuredAvgPower` / `POWER_BASIS_FIELDS[*].watts`, or `measuredJPerOutputToken` / `POWER_BASIS_FIELDS[*].energy` | +| `roles` | the selected scope (`all`, `prefill`, `decode`) | the other two worker pools | W: `measuredAvgPower`, `measuredPrefillAvgPower`, `measuredDecodeAvgPower`; J/out: `measuredJPerOutputToken`, `reconstructedPrefillJPerOutputToken`, `measuredDecodeJPerOutputToken` | + +Siblings exist only where the metric names the whole-deployment **average W/chip** or **J per +output token** (the only quantities every boundary and role publishes on one axis); role +energy additionally excludes the prefill J per input token scope. Elsewhere +`powerCompareVariants` returns nothing, the Compare select disables the option, and — when a +link arrives with an inapplicable mode — a hint (`measured-compare-hint`) says the comparison is +paused. The parameter is kept, so switching back to an applicable setting resumes it. + +Rendering (`ScatterGraph`): the series key is `scatterSeriesKey(point)` = +`_[-v-]` (`utils/point-identity.ts`), so rooflines, frontiers, +Optimal Only, line labels, overflow continuations and the perf ruler treat each sibling as its +own series; `parseScatterSeriesKey` recovers hardware, precision and variant wherever the key +was previously split on `_`. Siblings keep the hardware colour (overlay runs keep the run +colour) and take a per-variant `stroke-dasharray` (`powerVariantDash`); clone points render +at 0.6 opacity behind their base and carry `data-power-variant`. Line labels are placed per +hardware _and_ sibling: the base series keeps its plain hardware label, while a sibling appends +` · ` (`powerLineLabelSuffix`: TDP, All-in, PUE modeled, Prefill GPUs, …) and, when a +boundary is flat on a watts axis, its shared value (`B300 (SGLang) · TDP 1.2 kW`), so an exported +PNG explains its dashed lines without the legend; each pill carries `data-series-id` +(`::` for a sibling) and `data-power-variant`. The legend appends one +line-swatch row per series present (base first); rows toggle chart-local visibility +(`hiddenPowerVariants`, not in the URL) and hover-highlight that series across every hardware. +`scatterPointConfigId` includes the variant so a clone never replaces its base in a D3 join. +Tooltips add a "Series" line; on the energy axis a role clone also reports its share of the +reconstructed request energy. Table adds a "Series" column and CSV a trailing "Power Series" +column only while clones are present. Comparison clones are excluded from `bestSeriesPerSku`, +the power-tier counts, the legend points table, the availability panel (which reads +`selectionPoints`) and the date-comparison `GPUGraph`. Unofficial-run pills read `✕ ` (`getOverlayLineLabel`); the branch stays in the +legend and a short run tag (` · main`, ` · …-`) is appended only when several overlay +runs draw the same hardware. + +Figure 7's reconstruction lives in `utils/role-energy.ts`: schema-2 aggregate energy has one +numerator, so `J/out ÷ J/in` is the served input:output token ratio and +`prefill_joules_per_input_token × (J/out ÷ J/in)` is the prefill pool's energy per output +token; with `decode_joules_per_output_token` it sums back to the deployment's J/out when the +pool energies partition the total. `buildDerivedChartFields` emits it as +`reconstructedPrefillJPerOutputToken` for validated (`power_valid === 1`, schema 2) +disaggregated rows only; it is never a y-axis of its own. + +The Compare select tracks `inference_power_compare_changed { mode, family }`; legend rows +track `inference_power_compare_series_toggled { series, visible }`. + +## Power timeline (Display → Timeline) + +`y_measuredPowerTimeline` is the third value of the Measured Power **Display** control +(`watts` / `tdp` / `timeline`, `MeasuredPowerDisplay` in `measured-metric-config.ts`). Its +registry field aliases `measuredAvgPower`, so the point set, the availability panel, the Table +view and the share link are those of the measured average; only the chart body changes: +`ChartDisplay` renders `ui/PowerTimeline.tsx` instead of `ScatterGraph` when the resolved +config is `display: 'timeline'`. + +- **Join.** Each validated row's `power_audit.source` is `power_validation_.json`. + Two collectors publish the telemetry behind it (`utils/powerTimeline.ts`): + - single-node runners upload one `gpu_metrics_` CSV artifact per config + (`benchmark-tmpl.yml`), so the source names the artifact exactly + (`telemetryArtifactForPoint`); + - Slurm / Dynamo disaggregated runners upload one `power_audit_` bundle per + concurrency sweep whose `LOGS/power/samples.csv` (DCGM, `/` per device) + covers every concurrency. The API cuts it into one series per `power_validation_*.json` the + bundle contains (`components/gpu-power/power-audit-bundle.ts`: the validation file's + `selected_window` ± 60 s, roles from its `per_gpu_role`, manifest `expected_devices` as the + fallback) and labels each with that file name in `series.source`, which `joinPowerTimeline` + matches before falling back to the artifact name. + + Nothing is matched by hardware or concurrency. Rows whose telemetry is missing (expired + artifact, bundle over the download cap, another collector) are listed under the chart + (`data-testid="power-timeline-missing"`), never estimated. Trace keys are + `:` (`traceKeyForPoint`), unique per point in both collectors. + +- **Fetch.** One request per workflow run in the visible points + (`planPowerTimelineRequests`, at most `POWER_TIMELINE_MAX_RUNS`; a deep-linked trace's run + goes first, then `?unofficialrun=` overlay runs, then official runs — `prioritizeRun` / + `prioritizeRuns` — so an overlay the user asked for is never the run that gets dropped), + narrowed with + `prefix=` to the common RESULT_FILENAME prefix so a nightly sweep's other models are not + downloaded. `/api/gpu-metrics?series=power` first uses persisted telemetry and returns one-second per-GPU buckets + (`components/gpu-power/power-series.ts`, ~1 MB for a 25-config run instead of ~27 MB of + raw rows). Missing stored telemetry falls back to artifacts; database failures are explicit errors. + See [persistence and repair](./powerx-persistence-recovery.md) for cache freshness and coverage receipts. + With `series=power` the artifact fallback also downloads `power_audit_*` bundles whose name + shares the prefix (a bundle names the sweep, so the match runs both ways), reads only + `LOGS/power/samples.csv`, `LOGS/power/manifest.json` and the top-level + `power_validation_*.json` entries, and skips bundles above 256 MiB (the GB200 nw8 sweep is + 215 MB). A bundle-cut series carries `source` and `devices[] { id, role? }`; CSV series carry + neither. Runner timestamps are UTC wall clock and are parsed as such + (`parseTelemetryTimestampUtc`); `new Date('2026/09/12 20:19:57')` would read browser local + time and misplace the audit window. +- **Legend with overlays.** Once `?unofficialrun=` data is in, the chart reads + `localOfficialOverride`, so the timeline legend writes the unified selection + (`setUnifiedOverlaySelection`, `computeToggle` solo semantics) exactly as `ScatterGraph` + does; the context's `toggleHwType` alone would change nothing visible there. + +- **Drawing.** One trace per config, mean of its GPUs (legend switch: one line per GPU), + coloured by hardware for official rows and by `overlayRunColor(runIndex)` for + `?unofficialrun=` rows; legend toggles follow `activeHwTypes` / `activeOverlayHwTypes` + exactly as in `ScatterGraph`. The whole job is drawn faint and the `power_audit` window + emphasized (`data-segment="full" | "window"`). Rated TDP is a dashed reference per hardware; + the all-in provisioned line is an opt-in legend switch because it halves the traces' + vertical resolution. X axis: wall clock (UTC) when the visible traces come from one run, + otherwise seconds since each trace's start; both are a toolbar toggle. `c` labels sit + at the end of the emphasized segment. +- **Pools (Figure 1).** The legend switch _Prefill / decode pools_ (shown when a visible + trace carries worker roles) sums the board power of each role's GPUs + (`tracePools` / `sumPowerAt`) and draws one line per pool — prefill dashed `7 3`, decode + `2 3`, the same dashes as the `i_pcompare=roles` series — labelled `c · prefill` / + `· decode`, on a _GPU pool power (W)_ axis. Traces without roles draw their deployment total + as one `all` line so single-node configs stay comparable. Reference lines become pool size × + rated TDP per (hardware, pool size): roles of one hardware that hold the same GPU count share + one line labelled `prefill / decode ×16` (`data-pool` on `.power-reference` lists the roles, + `groupPoolsBySize`), and labels of lines at equal watts stack upward (`referenceLabelSlots`) + instead of overprinting; and the + all-in switch scales the same way. The tooltip names the pool, its summed watts against the + pool TDP, and the mean / min / max per GPU inside it. Pool mode and per-GPU lines are + mutually exclusive. +- **Deep link.** A pinned scatter tooltip on any metric of the measured family offers _View + power trace_ (`data-action="view-power-trace"`, official and `?unofficialrun=` points + alike). It switches the Display to Timeline in place and records the point's trace key in a + one-shot module store (`requestPowerTraceFocus`); the timeline consumes it on mount, dims + every other trace, switches to pool mode when the trace has roles, and shows a _Focused on …_ + chip (`data-testid="power-timeline-focus"`) with _Show all_ to clear. The link's `href` is + the current page with `i_metric=y_measuredPowerTimeline`, so open-in-new-tab lands on the + timeline (unfocused: the focus is a gesture, not URL state). +- **State.** Axis mode, line mode (mean / per GPU / pools), the all-in switch, the focused + trace and hover highlight are component state, not URL state: the share link is `i_metric=y_measuredPowerTimeline` plus the usual + scope, and a reader lands on the same defaults. +- **Analytics.** `inference_power_timeline_loaded { traces, missing, runs }`, + `inference_power_timeline_axis_changed { mode }`, + `inference_power_timeline_lines_changed { lines: 'mean' | 'gpu' | 'pool' }`, + `inference_power_timeline_utility_toggled { enabled }`, + `inference_power_trace_opened { hwKey, conc, overlay }` (scatter tooltip action), + `inference_power_timeline_focus_cleared`. + +## Tests + +- `lib/power-basis.test.ts` and `lib/chart-utils.test.ts` — the six boundary values through + the real builder, and the same fields on `?unofficialrun=` overlay rows. +- `gpu-power/power-series.test.ts`, `power-audit-bundle.test.ts` — one-second buckets with + `null` gaps, a partial pool is a gap rather than a lower sum, and bundle rows stay on their + host/device and role inside the padded window. +- `utils/powerTimeline.test.ts` — overlay runs are fetched ahead of official runs. +- `api/gpu-metrics/route*.test.ts` — the stored digest shape, DB-first serving, database + failure as 503, known missing hosts, and the retained-inventory recount before a CSV + fallback. +- `cypress/component/power-timeline.cy.tsx`, `power-compare.cy.tsx` — overlay-run colour and + the overlay hardware filter. diff --git a/docs/powerx-system-power.md b/docs/powerx-system-power.md index e1e1e5426..4eec8f70b 100644 --- a/docs/powerx-system-power.md +++ b/docs/powerx-system-power.md @@ -95,15 +95,20 @@ facility kW/GPU used to calculate capacity per GW. Consequently, revenue, compute expense, license fee, and profit scale together; profit margin does not change. Electricity expense is not recomputed separately. -This opt-in AgentX estimate requires validated schema-v2 telemetry and a -single-node chassis supported by the pinned model. Validated 1/2/4-GPU allocations -use the existing full-chassis extrapolation: fill an eight-GPU server with whole -replicas at the measured per-GPU power and throughput, then divide modeled facility -power by eight. This assumes replica co-location does not change performance or -power; it is not a measurement of a partly idle server. The chart, tooltip, and CSV -label every extrapolated estimate, including interpolation with one partial knot. -Unsupported GB200/GB300 chassis, multi-node layouts, allocations that cannot tile -eight GPUs, and missing/invalid measurements stay unavailable with distinct reasons. +This opt-in AgentX estimate requires validated schema-v2 telemetry and chassis +supported by the pinned model. Fully measured eight-GPU chassis are supported +on a single node, per measured worker host, or across an aggregate multinode +deployment without per-worker telemetry at the deployment-mean GPU power +(`topologyBasis: 'uniform-hosts'`; symmetric TP/PP/DP shards load each host alike). +Validated single-node 1/2/4-GPU allocations use full-chassis extrapolation: fill +an eight-GPU server with whole replicas at the measured per-GPU power and +throughput, then divide modeled facility power by eight. This assumes replica +co-location does not change performance or power; it is not a measurement of a +partly idle server. The chart, tooltip, and CSV label every extrapolated estimate, +including interpolation with one partial knot. Unsupported GB200/GB300 chassis, +partial multi-host allocations, disaggregated deployments without per-worker +telemetry, allocations that cannot tile eight GPUs, and missing/invalid +measurements stay unavailable with distinct reasons. The ordinary 8K/1K transformation keeps its existing admission policy. At an exact frontier point, use that point's modeled power. Between points, diff --git a/docs/state-ownership.md b/docs/state-ownership.md index d4922dae3..4c8ff69d5 100644 --- a/docs/state-ownership.md +++ b/docs/state-ownership.md @@ -113,12 +113,65 @@ them. Axis and presentation state: - selected x-axis and y-axis metrics, percentile, and effective x-axis mode +- the Measured controls (Boundary / Per / Scope / Statistic / Display / Unit) own no state: `measured-metric-config.ts` resolves each selection to the nearest registered metric key and writes it back to `selectedYAxisMetric`, so the power boundary rides on `i_metric` (there is deliberately no `i_pbasis`; see [PowerX Permanent View](./powerx-permanent-view.md)) +- the Measured Power Display value `timeline` (`y_measuredPowerTimeline`) swaps the chart body for `PowerTimeline`; its axis mode, per-GPU lines and all-in reference switch are component state and never enter the URL - token-revenue price source (`i_revenue`): normalized uncached/cached/output pricing or the selected model's live OpenRouter catalog prices - scale, optimal-point, label, contrast, legend, and overlay controls Display changes stay in this domain. A contrast or label toggle therefore does not notify filter-only or data-only consumers. +**`usePerfRulerStore`** (perf rulers, `i_rulers`) + +Completed Perf Rulers on the primary inference chart (`chart-0`) are provider state, +exposed through a dedicated `PerfRulerStoreContext` +(`packages/app/src/components/inference/perf-ruler-store.ts`) rather than the display domain: a +ruler commit would otherwise rerender every display consumer, and harnesses that mount +`InferenceContextsProvider` with static mock values would silently lose ruler +interactivity. The store is scoped to one chart id on purpose. The replay chart +(`replay-chart-0`) draws the same curve classes under the same provider, so a shared +store would render every ruler twice and let the replay's prune pass delete rulers the +main chart still shows; it and `GPUGraph` keep component-local state, which `ScatterGraph` +also falls back to when no store is present. + +`ScatterGraph` reads `state`/`setState` from the store, so the existing reducers, refs, +and draw passes are unchanged. The one thing that moved is the axis reset: +`usePerfRulerAxisReset` adjusts state during the chart's render, which is only legal for +the chart's own state, so for persisted rulers the same reset (either axis metric changes +→ rulers, draft, and pending share-link rulers are cleared) runs inside the provider. Its +key, `persistedPerfRulerAxisKey`, describes the graph `ChartDisplay` renders as `chart-0`, +not `graphs[0]`: `graphs` is always `[interactivity, e2e]` and ChartDisplay shows the e2e +graph for every non-interactivity x mode, so the key picks the graph by `selectedXAxisMode` +(as `bestHwTypes` does), uses its `x_scale_field` (which carries the percentile), and folds +the mode itself in because the derived agentic modes override the e2e graph's x field inside +ChartDisplay only. The first chart definition (null → key) and a reload of the same axes +(key → null → key) are not axis changes, so share-link rulers survive the load. + +Restoring from a link is a two-phase commit because of a load race. Rulers parsed from +`i_rulers` start as `pending`; data, `i_gpus`, comparison dates, and `?unofficialrun=` +overlays all arrive after the chart's first draw, and the chart prunes any committed +ruler whose curve path is absent from the DOM. The chart therefore commits a pending +ruler only once BOTH of its curve paths exist (hidden-at-opacity-0 counts as present, as +for prune), clamping the iso-x to the pair's overlap through the drawn paths. The commit +runs inside the perf-ruler D3 layer's draw pass (`drawPerfRuler`), not in a React effect: +the chart's first draw happens in a `D3Chart`-local re-render after its container is +measured, which re-renders nothing in `ScatterGraph`, so an effect keyed on the chart's +props could miss the first draw and leave resolvable rulers pending for the session. No +commit happens while the ruler mode is off (the power envelope forces it off on +non-measured axes; the mode-off effect discards pending rulers instead). Rulers whose +curves never appear stay pending — invisible and unserialized — until an axis change, +toggle-off, or clear discards them; this is a deliberate choice over a "data settled" +readiness predicate (fragile across the async order of availability, benchmark, overlay, +and derived-metric fetches). The visible cost is that a link whose curves the recipient's +view lacks lands with the ruler switch on and nothing drawn, and a later legend change on +the same axes can surface the ruler. A non-empty restore switches the ruler mode on for +that chart instance (rulers only render in mode) and fires +`interactivity_perf_ruler_shared_load` once per opened link, with `count` = the number of +rulers the link carried, the first time any of them renders — never per commit batch, so +the event counts links, not curve arrivals. Because the URL is read in a `useState` +initialiser, a retained-provider tab switch onto `/inference?i_rulers=…` does not +re-import the param — the same limitation as the other `i_*` toggles. + **`useInferenceActions`** All inference commands and setters. The context value and every exposed action keep @@ -360,3 +413,24 @@ Dashboard scope membership is declared by `shareParamScopes` in `packages/app/src/lib/dashboard-routes.ts`. Tests enforce completeness and route-specific share behavior, so this document deliberately does not duplicate a manually maintained parameter table. + +One entry needs a note on its encoding: `i_rulers` (inference scope, default `''`) is +`serializePerfRulers` output — `isoX|curveA|curveB` per ruler joined by `;`, where the +curve ids are the rendered roofline path identity classes (`roofline-_`, +`overlay-roofline-__run`, optionally `__`) and the +iso-x is in DATA space rounded to four significant digits. `parsePerfRulers` only accepts +curve ids of that identity-class shape — the shared `roofline-path` marker class or any +other zoom-group node would match many paths and draw a ruler between arbitrary curves. +Only committed rulers are serialized (never the draft or still-pending link rulers), and +the chart prunes them against the rendered curves, so a link written from one data set +degrades to fewer rulers, never to an error. Overlay ids depend on the run order in `unofficialruns`, which the share +link already carries. + +`i_pcompare` (inference scope, default `''`) is the power comparison mode, `boundaries` or +`roles`, owned by `InferenceProvider` as `powerCompare` (display context) with +`setPowerCompare` (actions). It is written as `''` for `none`. It is deliberately NOT cleared +when the metric changes to one without a common axis for the siblings: the chart then draws +the metric alone and the Measured controls show a hint, so the link's intent survives a +detour through another setting. Which comparison rows a reader hid from the legend is +`ScatterGraph` component state (`hiddenPowerVariants`), never shared — see +[PowerX Permanent View](./powerx-permanent-view.md#comparison-series-i_pcompare). diff --git a/packages/app/cypress/component/inference-chart-controls.cy.tsx b/packages/app/cypress/component/inference-chart-controls.cy.tsx index ded6bd9ba..4445d303b 100644 --- a/packages/app/cypress/component/inference-chart-controls.cy.tsx +++ b/packages/app/cypress/component/inference-chart-controls.cy.tsx @@ -469,8 +469,10 @@ describe('Inference ChartControls grouped measured metrics', () => { cy.get(`input[aria-label="${searchLabel}"]`).type(group); cy.get('[data-select-option][data-value^="y_measured"]').should(($options) => { const values = [...$options].map((option) => option.dataset.value); - expect(values).to.have.length(13); - expect(new Set(values).size).to.equal(13); + // Thirteen measured axes plus the Timeline display of measured power. + expect(values).to.have.length(14); + expect(new Set(values).size).to.equal(14); + expect(values).to.include('y_measuredPowerTimeline'); }); cy.get(`input[aria-label="${searchLabel}"]`).clear().type(power); cy.get('[data-select-option]') diff --git a/packages/app/cypress/component/power-compare.cy.tsx b/packages/app/cypress/component/power-compare.cy.tsx new file mode 100644 index 000000000..f2a51b485 --- /dev/null +++ b/packages/app/cypress/component/power-compare.cy.tsx @@ -0,0 +1,183 @@ +import { PathnameContext } from 'next/dist/shared/lib/hooks-client-context.shared-runtime'; + +import ScatterGraph from '@/components/inference/ui/ScatterGraph'; +import type { InferenceData } from '@/components/inference/types'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; +import { Precision } from '@/lib/data-mappings'; +import { overlayRunColor } from '@/lib/overlay-run-style'; + +import { + createMockChartDefinition, + createMockHardwareConfig, + createMockInferenceData, +} from '../support/mock-data'; +import { mountWithProviders } from '../support/test-utils'; + +// Power comparison series (`i_pcompare`) on `?unofficialrun=` overlays: role +// clones draw in the overlay run colour with the role dash, and their legend +// rows toggle them. + +const OVERLAY_RUN_ID = 31415926535; +const OVERLAY_RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${OVERLAY_RUN_ID}`; +const hwConfig = createMockHardwareConfig(); +const chartDefinition = createMockChartDefinition({ + chartType: 'interactivity', + y_measuredAvgPower: 'measuredAvgPower.y', + y_measuredAvgPower_roofline: 'lower_left', +}); + +const metric = (y: number) => ({ y, roof: false }); + +/** + * Three measured points with both roles available. Power falls as x rises so + * every point sits on the upper power envelope the chart draws for watt axes. + */ +function measuredCurve(hwKey: string, run_url?: string): InferenceData[] { + return [ + [8, 700, 840, 500], + [16, 600, 820, 450], + [32, 500, 800, 400], + ].map(([x, measured, prefill, decode]) => + createMockInferenceData({ + hwKey, + x, + conc: x, + y: measured, + precision: Precision.FP4, + run_url, + disagg: true, + measuredAvgPower: metric(measured), + measuredPrefillAvgPower: metric(prefill), + measuredDecodeAvgPower: metric(decode), + }), + ); +} + +function mountCompare(data: InferenceData[], overlay: InferenceData[]) { + mountWithProviders( + +
+ +
+
, + { + inference: { + selectedYAxisMetric: 'y_measuredAvgPower', + hardwareConfig: hwConfig, + activeHwTypes: new Set(['b200', 'h100']), + hwTypesWithData: new Set(['b200', 'h100']), + selectedPrecisions: [Precision.FP4], + hideNonOptimal: false, + showLineLabels: false, + }, + unofficial: { + isUnofficialRun: true, + activeOverlayHwTypes: new Set(['h100']), + allOverlayHwTypes: new Set(['h100']), + runIndexByUrl: { [OVERLAY_RUN_URL]: 0, [String(OVERLAY_RUN_ID)]: 0 }, + unofficialRunInfos: [ + { + id: OVERLAY_RUN_ID, + name: 'powerx-compare', + branch: 'powerx-compare', + sha: 'abc000', + createdAt: '2026-09-01T00:00:00Z', + url: OVERLAY_RUN_URL, + conclusion: 'success', + status: 'completed', + isNonMainBranch: true, + }, + ], + }, + }, + ); +} + +const svg = '#power-compare-test svg'; +const legend = '#power-compare-test [data-testid="chart-legend"]'; + +describe('ScatterGraph power comparison series', () => { + beforeEach(() => { + cy.on('uncaught:exception', (error) => { + if (error.message.includes('ResizeObserver loop')) return false; + }); + }); + + it('draws role siblings for ?unofficialrun= overlays in the run colour with the role dash', () => { + const official = expandPowerCompareSeries(measuredCurve('b200'), 'y_measuredAvgPower', 'roles'); + const overlay = expandPowerCompareSeries( + measuredCurve('h100', OVERLAY_RUN_URL), + 'y_measuredAvgPower', + 'roles', + ); + mountCompare(official, overlay); + + cy.get(`${svg} .roofline-path[data-power-variant="prefill"]`).should( + 'have.attr', + 'stroke-dasharray', + '7 3', + ); + cy.get(`${svg} .unofficial-overlay-pt`).should('have.length', 9); + cy.get(`${svg} .overlay-roofline-path[data-power-variant="decode"]`) + .should('have.length', 1) + .and('have.attr', 'stroke', overlayRunColor(0)) + .and('have.attr', 'stroke-dasharray', '2 3'); + cy.get(`${svg} .overlay-roofline-path:not([data-power-variant])`).should('have.length', 1); + cy.get(legend).within(() => { + cy.contains('All GPUs').should('exist'); + cy.contains('Prefill GPUs').should('exist'); + cy.contains('Decode GPUs').click(); + }); + cy.get(`${svg} .overlay-roofline-path[data-power-variant="decode"]`).should( + 'have.css', + 'opacity', + '0', + ); + cy.get(`${svg} .unofficial-overlay-pt`).then(($points) => { + const hidden = [...$points].filter((point) => getComputedStyle(point).opacity === '0'); + expect(hidden).to.have.length(3); + }); + }); + + it('keeps the base series lit when its legend row is hovered', () => { + const official = expandPowerCompareSeries(measuredCurve('b200'), 'y_measuredAvgPower', 'roles'); + const overlay = expandPowerCompareSeries( + measuredCurve('h100', OVERLAY_RUN_URL), + 'y_measuredAvgPower', + 'roles', + ); + mountCompare(official, overlay); + + cy.get(`${svg} .roofline-path[data-power-variant="prefill"]`).should('exist'); + // Base points and rooflines carry no variant id (only role siblings are + // cloned), so the base row has to map back onto them. + cy.get(legend).contains('label', 'All GPUs').trigger('mouseover'); + // Wait for the siblings to settle first: opacity transitions over 150 ms, + // so a base check taken at once would still read the pre-hover value. + cy.get(`${svg} .roofline-path[data-power-variant="prefill"]`).should( + 'have.css', + 'opacity', + '0.15', + ); + cy.get(`${svg} .roofline-path:not([data-power-variant])`).should('have.css', 'opacity', '1'); + cy.get(legend).contains('label', 'All GPUs').trigger('mouseout'); + cy.get(`${svg} .roofline-path[data-power-variant="prefill"]`).should( + 'have.css', + 'opacity', + '1', + ); + }); +}); diff --git a/packages/app/cypress/component/power-telemetry.cy.tsx b/packages/app/cypress/component/power-telemetry.cy.tsx new file mode 100644 index 000000000..332a5fde4 --- /dev/null +++ b/packages/app/cypress/component/power-telemetry.cy.tsx @@ -0,0 +1,205 @@ +import { QueryClient, QueryClientProvider, useQuery } from '@tanstack/react-query'; +import { PathnameContext } from 'next/dist/shared/lib/hooks-client-context.shared-runtime'; +import { useState } from 'react'; + +import { TelemetryDisplayControls } from '@/components/gpu-power/TelemetryDisplayControls'; +import { + DEFAULT_TELEMETRY_DISPLAY, + type TelemetryDisplayState, +} from '@/components/gpu-power/telemetry-smoothing'; +import { PowerTelemetryView } from '@/components/inference/agentic-point/power-telemetry-view'; +import type { GpuMetricsPointPayload, GpuMetricSeries } from '@/hooks/api/use-gpu-metrics-point'; +import { registerAnalyticsClient } from '@/lib/analytics'; + +const ID = 206887; +const endpoint = `/api/v1/gpu-metrics-point?id=${ID}`; +const queryKey = ['gpu-metrics-point', ID]; +const chart = '[data-testid="power-telemetry-chart"]'; + +function series(id: number, host: string, base: number): GpuMetricSeries { + const data = [0, 1, 2].flatMap((second) => + [0, 1].map((index) => ({ + timestamp: `2026-09-21T00:00:0${second}Z`, + index, + power: base + index * 100 + second * 10, + temperature: 40 + index + second, + })), + ); + return { + id, + artifactName: 'gpu_metrics_qwen35_b200', + configKey: 'qwen35_b200_tp8', + fileName: `${host}/gpu_metrics.csv`, + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: data.length, + gpuCount: 2, + startedAt: data[0].timestamp, + endedAt: data.at(-1)!.timestamp, + sidecars: {}, + benchmarkResultIds: [ID], + stats: [0, 1].map((gpuIndex) => ({ + gpuIndex, + metric: 'power_w', + count: 3, + min: base + gpuIndex * 100, + max: base + gpuIndex * 100 + 20, + mean: base + gpuIndex * 100 + 10, + median: base + gpuIndex * 100 + 10, + p95: base + gpuIndex * 100 + 19, + p99: base + gpuIndex * 100 + 19.8, + stddev: Math.sqrt(200 / 3), + })), + data, + }; +} + +const payload: GpuMetricsPointPayload = { + benchmarkResultId: ID, + series: [series(1, 'host-a', 500), series(2, 'host-b', 700)], +}; + +function QueryStatus() { + const { status } = useQuery({ queryKey, enabled: false }); + return {status}; +} + +function mountPoint(path = '/inference/agentic/206887') { + const client = new QueryClient({ defaultOptions: { queries: { retry: false, gcTime: 0 } } }); + cy.mount( + + +
+ + +
+
+
, + ); + return client; +} + +function Controls() { + const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); + return ( + <> + + {JSON.stringify(display)} + + ); +} + +describe('PowerX telemetry interactions', () => { + beforeEach(() => { + registerAnalyticsClient({ capture: cy.stub().as('capture') }); + }); + + it('changes display, window and chip aggregation independently and tracks each action', () => { + cy.mount(); + cy.get('[data-testid="display-mode-points"]').click(); + cy.get('#display-window').should('not.exist'); + cy.get('[data-testid="display-series-mean"]').click(); + cy.get('[data-testid="display-mode-rolling"]').click(); + cy.get('#display-window').click(); + cy.get('[role="option"]').contains('60 s').click(); + cy.get('output').should('have.text', '{"mode":"rolling","windowS":60,"series":"mean"}'); + cy.get('@capture').should('have.been.calledWith', 'power_test_display_mode_changed', { + mode: 'points', + }); + cy.get('@capture').should('have.been.calledWith', 'power_test_series_mode_changed', { + series: 'mean', + }); + cy.get('@capture').should('have.been.calledWith', 'power_test_smoothing_window_changed', { + windowS: 60, + }); + cy.get('@capture').its('callCount').should('eq', 4); + }); + + it('shows loading, distinguishes an initial failure from missing data, and retries', () => { + let failed = true; + cy.intercept('GET', endpoint, (request) => { + request.reply(failed ? { statusCode: 503, delay: 150, body: {} } : { body: payload }); + }).as('telemetry'); + mountPoint(); + cy.get('[data-testid="power-telemetry-loading"]').should('contain.text', 'Loading PowerX'); + cy.wait('@telemetry'); + cy.get('[data-testid="power-telemetry-query-error"]').should('contain.text', 'Failed to load'); + cy.get('[data-testid="power-telemetry-missing"]').should('not.exist'); + cy.then(() => { + failed = false; + }); + cy.contains('button', 'Retry').click(); + cy.wait('@telemetry'); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.get('@capture').should( + 'have.been.calledWith', + 'inference_agentic_power_telemetry_retry_clicked', + ); + }); + + it('shows a genuine missing point without offering an error retry', () => { + cy.intercept('GET', endpoint, { statusCode: 404, body: {} }); + mountPoint(); + cy.get('[data-testid="power-telemetry-missing"]').should('contain.text', `#${ID}`); + cy.get('[data-testid="power-telemetry-query-error"]').should('not.exist'); + cy.get(chart).should('not.exist'); + }); + + it('preserves a rendered chart when a background refetch fails', () => { + let failed = false; + cy.intercept('GET', endpoint, (request) => { + request.reply(failed ? { statusCode: 503, body: {} } : { body: payload }); + }).as('telemetry'); + const client = mountPoint(); + cy.wait('@telemetry'); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.then(() => { + failed = true; + return client.refetchQueries({ queryKey }); + }); + cy.wait('@telemetry'); + cy.get('[data-testid="query-status"]').should('have.text', 'error'); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.get('[data-testid="power-telemetry-query-error"]').should('not.exist'); + }); + + for (const [locale, width] of [ + ['en', 1280], + ['zh', 390], + ] as const) { + it(`keeps chip filters scoped to a host and renders ${locale} at ${width}px`, () => { + cy.viewport(width, 900); + cy.intercept('GET', endpoint, { body: payload }); + mountPoint(`${locale === 'zh' ? '/zh' : ''}/inference/agentic/${ID}`); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.get('[data-testid="power-telemetry-sample-count"]').should('have.text', '6'); + cy.get('table tbody tr').first().find('td').eq(4).should('have.text', '510.0'); + cy.get('[data-testid="chart-legend"]') + .contains(locale === 'zh' ? '芯片 0' : 'Chip 0') + .click(); + cy.get(chart).find('svg .point').should('have.length', 3); + cy.get('#power-telemetry-series-select').click(); + cy.get('[role="option"]').contains('host-b').click(); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.get('table tbody tr').first().find('td').eq(4).should('have.text', '710.0'); + cy.get('[data-testid="power-telemetry-display-mode-points"]').click(); + cy.get('[data-testid="power-telemetry-display-series-mean"]').click(); + cy.get(chart).find('svg .point').should('have.length', 3); + cy.get('[data-testid="power-telemetry-view"]').should(($view) => { + expect($view[0].scrollWidth).to.be.at.most($view[0].clientWidth + 1); + }); + cy.get('[data-testid="power-telemetry-metric-select"]').should( + 'contain.text', + locale === 'zh' ? '功耗' : 'Power', + ); + cy.get('[data-testid="power-telemetry-view"]').screenshot( + `power-telemetry-${locale}-${width}`, + ); + }); + } +}); diff --git a/packages/app/cypress/component/power-timeline.cy.tsx b/packages/app/cypress/component/power-timeline.cy.tsx new file mode 100644 index 000000000..7df453db1 --- /dev/null +++ b/packages/app/cypress/component/power-timeline.cy.tsx @@ -0,0 +1,244 @@ +import { PathnameContext } from 'next/dist/shared/lib/hooks-client-context.shared-runtime'; +import { useState } from 'react'; + +import type { GpuPowerSeries, GpuPowerSeriesResponse } from '@/components/gpu-power/power-series'; +import PowerTimeline from '@/components/inference/ui/PowerTimeline'; +import type { InferenceData } from '@/components/inference/types'; +import { Model, Precision, Sequence } from '@/lib/data-mappings'; +import { overlayRunColor } from '@/lib/overlay-run-style'; + +import { + createMockHardwareConfig, + createMockInferenceData, + createMockUnofficialRunContext, +} from '../support/mock-data'; +import { mountWithProviders } from '../support/test-utils'; + +// PowerTimeline joins chart points to `gpu_metrics_` artifacts +// by the `power_audit.source` file name and draws one trace per config. Overlay +// runs keep their run colour and follow the overlay hardware filter. + +const RUN_ID = '34716669498'; +const RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${RUN_ID}`; +const OVERLAY_RUN_ID = '31415926535'; +const OVERLAY_RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${OVERLAY_RUN_ID}`; +const SECOND_RUN_ID = '34716669499'; +const SECOND_RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${SECOND_RUN_ID}`; +const START_MS = Date.UTC(2026, 8, 12, 20, 20, 0); +const hwConfig = createMockHardwareConfig(); +const HW_TYPES = new Set(['b200', 'h100']); + +const resultName = (hardware: string, conc: number) => + `dsv4_8k1k_fp4_sglang_tp8-pp1-dcp1-pcp1-ep1-dpafalse_disagg-false_spec-none_conc${conc}_${hardware}-host-0123456789abcdef0123`; +const WINDOW = { + // Window covers the last 20 s of a 60 s job. + window_start_unix: (START_MS + 40_000) / 1000, + window_end_unix: (START_MS + 60_000) / 1000, +}; + +function measuredPoint( + hwKey: string, + conc: number, + watts: number, + overrides: Partial = {}, +): InferenceData { + return createMockInferenceData({ + hwKey, + conc, + tp: 8, + x: conc, + y: watts, + precision: Precision.FP4, + run_url: RUN_URL, + measuredAvgPower: { y: watts, roof: true }, + measuredPowerTimeline: { y: watts, roof: true }, + power_audit: { + source: `power_validation_${resultName(hwKey.split('_')[0], conc)}.json`, + ...WINDOW, + }, + ...overrides, + }); +} + +/** 61 one-second buckets: idle 200 W, ramps to `peak` inside the window. */ +function series(hwKey: string, conc: number, peak: number, gpus = [0, 1]): GpuPowerSeries { + const t = Array.from({ length: 61 }, (_, i) => i); + return { + artifact: `gpu_metrics_${resultName(hwKey.split('_')[0], conc)}`, + startMs: START_MS, + bucketSeconds: 1, + gpus, + t, + power: gpus.map((gpu) => t.map((second) => (second >= 40 ? peak + gpu * 10 : 200 + gpu))), + }; +} + +const response: GpuPowerSeriesResponse = { + runInfo: { + id: Number(RUN_ID), + name: 'Run Sweep', + branch: 'main', + sha: 'abc123', + createdAt: '2026-09-12T20:00:00Z', + url: RUN_URL, + conclusion: 'success', + status: 'completed', + }, + series: [series('b200', 16, 700), series('b200', 64, 900)], +}; + +const Y_LABEL = 'Measured Average Power per Chip over Time (W)'; + +function providerOverrides( + unofficial: Parameters[0] = {}, +): Parameters[1] { + return { + inference: { + selectedModel: Model.DeepSeek_V4_Pro, + selectedSequence: Sequence.EightK_OneK, + selectedYAxisMetric: 'y_measuredPowerTimeline', + hardwareConfig: hwConfig, + activeHwTypes: new Set(HW_TYPES), + hwTypesWithData: new Set(HW_TYPES), + }, + unofficial, + }; +} + +function mountTimeline( + data: InferenceData[], + options: { + overlay?: Parameters[0]['overlayData']; + unofficial?: Parameters[0]; + } = {}, +) { + mountWithProviders( + +
+ +
+
, + providerOverrides(options.unofficial), + ); +} + +/** One measured run at mount; a button adds a second run to the same plot. */ +function GrowingTimeline() { + const [data, setData] = useState(() => [measuredPoint('b200', 16, 700)]); + return ( + + +
+ +
+
+ ); +} + +const svg = () => cy.get('[data-testid="power-timeline-chart-svg"]'); + +describe('PowerTimeline', () => { + beforeEach(() => { + cy.on('uncaught:exception', (error) => { + if (error.message.includes('ResizeObserver loop')) return false; + }); + }); + + it('colours overlay-run traces by run and honours the overlay hardware filter', () => { + const overlayPoint = measuredPoint('h200', 16, 500, { + run_url: OVERLAY_RUN_URL, + power_audit: { + source: `power_validation_${resultName('h200', 16)}.json`, + ...WINDOW, + }, + }); + const overlayResponse: GpuPowerSeriesResponse = { + runInfo: { ...response.runInfo, id: Number(OVERLAY_RUN_ID), url: OVERLAY_RUN_URL }, + series: [series('h200', 16, 500)], + }; + cy.intercept('POST', `/api/gpu-metrics?runId=${RUN_ID}*`, { body: response }).as('official'); + cy.intercept('POST', `/api/gpu-metrics?runId=${OVERLAY_RUN_ID}*`, { + body: overlayResponse, + }).as('overlay'); + mountTimeline([measuredPoint('b200', 16, 700)], { + overlay: { + data: [overlayPoint], + hardwareConfig: hwConfig, + label: 'powerx-timeline', + runUrl: OVERLAY_RUN_URL, + }, + unofficial: createMockUnofficialRunContext({ + isUnofficialRun: true, + unofficialRunInfos: [ + { + id: Number(OVERLAY_RUN_ID), + name: 'powerx-timeline', + branch: 'powerx-timeline', + sha: 'abc000', + createdAt: '2026-09-12T00:00:00Z', + url: OVERLAY_RUN_URL, + conclusion: 'success', + status: 'completed', + isNonMainBranch: true, + }, + ], + runIndexByUrl: { [OVERLAY_RUN_URL]: 0, [OVERLAY_RUN_ID]: 0 }, + activeOverlayHwTypes: new Set(['h200']), + }), + }); + cy.wait(['@official', '@overlay']); + + svg().within(() => { + cy.get('path.power-trace[data-run-index="0"][data-segment="window"]') + .should('have.length', 1) + .and('have.attr', 'stroke', overlayRunColor(0)); + cy.get('path.power-trace[data-hw="b200"][data-segment="window"]').should('have.length', 1); + // Two runs: the axis defaults to elapsed time so traces overlap by phase. + cy.get('text').contains('Time since telemetry start').should('exist'); + // Reference lines cover both hardware SKUs. + cy.get('.power-reference[data-reference="tdp"]').should('have.length', 2); + }); + cy.get('[data-testid="chart-legend"]').should('contain.text', '✕ powerx-timeline'); + }); + + it('joins a run that arrives after mount without a hook-shape warning', () => { + const secondResponse: GpuPowerSeriesResponse = { + runInfo: { ...response.runInfo, id: Number(SECOND_RUN_ID), url: SECOND_RUN_URL }, + series: [series('h100', 16, 500)], + }; + cy.intercept('POST', `/api/gpu-metrics?runId=${RUN_ID}*`, { body: response }).as('first'); + cy.intercept('POST', `/api/gpu-metrics?runId=${SECOND_RUN_ID}*`, { + body: secondResponse, + }).as('second'); + cy.stub(console, 'error').as('consoleError'); + mountWithProviders(, providerOverrides()); + cy.wait('@first'); + svg().find('path.power-trace[data-segment="window"]').should('have.length', 1); + + cy.get('[data-testid="add-run"]').click(); + cy.wait('@second'); + svg().find('path.power-trace[data-segment="window"]').should('have.length', 2); + // React logs this when a memo's dependency array changes length between + // renders; one query per run used to be spread into that array. + cy.get('@consoleError').then((stub) => { + const calls = (stub as unknown as { args: unknown[][] }).args; + const shapeWarnings = calls.filter((args) => + args.some((a) => typeof a === 'string' && a.includes('changed size between renders')), + ); + expect(shapeWarnings, JSON.stringify(shapeWarnings)).to.have.length(0); + }); + }); +}); diff --git a/packages/app/cypress/component/scatter-graph.cy.tsx b/packages/app/cypress/component/scatter-graph.cy.tsx index 0f157fea1..fac02fad5 100644 --- a/packages/app/cypress/component/scatter-graph.cy.tsx +++ b/packages/app/cypress/component/scatter-graph.cy.tsx @@ -925,14 +925,19 @@ describe('ScatterGraph', () => { cy.get('#test-scatter-overlay-labels svg .line-label') .filter('[data-line-key]:not([data-line-key^="overlay-"])') .should('have.length.greaterThan', 0); - // The exact branch that crashed the production page remains visible in the - // overlay line label and legend after ScatterGraph's render-time updates. - cy.get('#test-scatter-overlay-labels svg .line-label[data-line-key^="overlay-"]') - .find('text') - .should('contain.text', runBranch); + // The pill names the hardware behind the ✕ marker, parsed like an official + // pill; the long branch that crashed the production page stays in the legend. + cy.get('#test-scatter-overlay-labels svg .line-label[data-line-key^="overlay-"] .ll-text') + .should('have.text', '✕ B200 (TRTLLM)') + .and('not.contain.text', runBranch); cy.get( '#test-scatter-overlay-labels svg .line-label[data-line-key^="overlay-"] .ll-gpu', - ).should('not.exist'); + ).should('have.text', 'B200'); + // b200_trt is active only in the overlay legend (official rows: h100), so + // the overlay pill must stay visible after the filter-sync effect. + cy.get('#test-scatter-overlay-labels svg .line-label[data-line-key^="overlay-"]') + .should('have.attr', 'data-visible', '1') + .and('have.css', 'opacity', '1'); cy.get('#test-scatter-overlay-labels [data-testid="chart-legend"]').should( 'contain.text', runBranch, @@ -1094,7 +1099,7 @@ describe('ScatterGraph', () => { cy.get('#test-scatter-singleton-overlay-label svg .line-label[data-line-key^="overlay-"]') .should('have.length', 1) .find('text') - .should('contain.text', 'tileRT'); + .should('have.text', '✕ B200 (TRTLLM)'); cy.get('#test-scatter-singleton-overlay-label svg').then(($svg) => { const svg = $svg[0]; diff --git a/packages/app/cypress/support/mock-data.ts b/packages/app/cypress/support/mock-data.ts index 413f38fab..733d04e49 100644 --- a/packages/app/cypress/support/mock-data.ts +++ b/packages/app/cypress/support/mock-data.ts @@ -232,6 +232,8 @@ export function createMockInferenceContextValues( setSelectedXAxisMode: namedStub('setSelectedXAxisMode'), scaleType: 'auto', setScaleType: namedStub('setScaleType'), + powerCompare: 'none' as const, + setPowerCompare: namedStub('setPowerCompare'), quickFilters: { vendors: [], frameworks: [], deployment: [], spec: [], power: [] }, availableQuickFilters: { vendors: [], frameworks: [], deployment: [], spec: [], power: [] }, setQuickFilterVendors: namedStub('setQuickFilterVendors'), diff --git a/packages/app/src/components/calculator/profit-power.ts b/packages/app/src/components/calculator/profit-power.ts index 089abb453..1c6bc2a1b 100644 --- a/packages/app/src/components/calculator/profit-power.ts +++ b/packages/app/src/components/calculator/profit-power.ts @@ -40,8 +40,12 @@ function planningPower(point: GPUDataPoint): PlanningPower { } } } - // Full-chassis planning requires whole replicas to fit on one eight-GPU host. - if (estimate.topologyBasis !== 'single-node' || 8 % estimate.gpuCount !== 0) + // Partial allocations must tile one host; fully measured multi-host estimates + // retain their validated worker-hosts or uniform-hosts topology. + if ( + estimate.chassisBasis === 'extrapolated' && + (estimate.topologyBasis !== 'single-node' || 8 % estimate.gpuCount !== 0) + ) return { reason: 'unsupported-power-topology' }; return { kwPerGpu: (estimate.deploymentFacilityWatts / estimate.gpuCount / 1000) * 1.1, diff --git a/packages/app/src/components/inference/InferenceContext.tsx b/packages/app/src/components/inference/InferenceContext.tsx index 036f22700..206c9fd47 100644 --- a/packages/app/src/components/inference/InferenceContext.tsx +++ b/packages/app/src/components/inference/InferenceContext.tsx @@ -37,6 +37,7 @@ import type { InferenceDataContextType, InferenceDisplayContextType, InferenceFiltersContextType, + PowerCompare, TokenRevenuePriceSource, } from '@/components/inference/types'; import { resolveMetricConfigKey } from '@/components/inference/metric-registry'; @@ -56,6 +57,14 @@ import { useUrlStateSync, } from '@/hooks/useChartContext'; import { useUrlState } from '@/hooks/useUrlState'; +import { serializePerfRulers } from '@/lib/d3-chart/layers/perf-ruler'; +import { parsePowerCompare } from '@/components/inference/utils/power-compare'; +import { + PERSISTED_PERF_RULER_CHART_ID, + PerfRulerStoreContext, + persistedPerfRulerAxisKey, + usePerfRulerStoreValue, +} from '@/components/inference/perf-ruler-store'; import { useParetoHighlightToggle } from './hooks/useParetoHighlightToggle'; import { useOpenRouterPricing } from '@/hooks/api/use-openrouter-pricing'; import { DEFAULT_Y_AXIS_METRIC } from '@/lib/url-state'; @@ -483,6 +492,12 @@ export function InferenceProvider({ const [scaleType, setScaleType] = useState<'auto' | 'linear' | 'log'>( () => (getUrlParam('i_scale') as 'auto' | 'linear' | 'log') || 'auto', ); + // Comparison series on a gated power metric (`i_pcompare`). Kept while the + // metric changes: a key without a common axis simply yields no siblings, and + // the Measured controls say so, so a link's intent survives a detour. + const [powerCompare, setPowerCompare] = useState(() => + parsePowerCompare(getUrlParam('i_pcompare')), + ); // ── Quick filters (vendor / framework / deployment / mtp-stp / power tier) ── // Coarse pre-filters applied to the point set. Empty = no constraint. @@ -770,6 +785,7 @@ export function InferenceProvider({ !isUnofficialRun && !hasExplicitRunSelection && selectedRunDateRev === 0, + powerCompare, ); // For GPU comparison date picker — use shared availability data from global filters @@ -1035,6 +1051,21 @@ export function InferenceProvider({ const refreshing = !availabilityError && chartDataRefreshing; const error = availabilityError || workflowError || chartDataError; + // ── Perf rulers (persisted chart) ──────────────────────────────────────── + // The axis identity follows the graph ChartDisplay renders as `chart-0` + // (picked by x mode, like `bestHwTypes` below), so an x-mode switch that + // swaps the rendered chart or its x units clears the rulers the same way + // the chart's own `usePerfRulerAxisReset` does for local state. + const perfRulerStore = usePerfRulerStoreValue( + PERSISTED_PERF_RULER_CHART_ID, + getUrlParam('i_rulers'), + persistedPerfRulerAxisKey(graphs, selectedXAxisMode, selectedYAxisMetric), + ); + const iRulersStr = useMemo( + () => serializePerfRulers(perfRulerStore.state), + [perfRulerStore.state], + ); + // ── Toggle sets ─────────────────────────────────────────────────────────── const { @@ -1603,6 +1634,8 @@ export function InferenceProvider({ i_disagg: quickFilterDeployment.join(','), i_spec: quickFilterSpec.join(','), i_power: quickFilterPower.join(','), + i_rulers: iRulersStr, + i_pcompare: powerCompare === 'none' ? '' : powerCompare, }, [ selectedYAxisMetric, @@ -1634,6 +1667,8 @@ export function InferenceProvider({ quickFilterDeployment, quickFilterSpec, quickFilterPower, + iRulersStr, + powerCompare, ], ); @@ -1849,6 +1884,7 @@ export function InferenceProvider({ selectedE2eXAxisMetric, selectedXAxisMode, scaleType, + powerCompare, isLegendExpanded, hideNonOptimal, showAllMeasurements, @@ -1874,6 +1910,7 @@ export function InferenceProvider({ selectedE2eXAxisMetric, selectedXAxisMode, scaleType, + powerCompare, isLegendExpanded, hideNonOptimal, showAllMeasurements, @@ -1908,6 +1945,7 @@ export function InferenceProvider({ setSelectedXAxisMetric, setSelectedXAxisMode: handleSetXAxisMode, setScaleType, + setPowerCompare, setQuickFilterVendors, setQuickFilterFrameworks, setQuickFilterDeployment, @@ -1944,7 +1982,9 @@ export function InferenceProvider({ display={displayValue} actions={actionsValue} > - {children} + + {children} + [ { value: 'point', label: t.perPoint, testId: 'detail-view-point' }, { value: 'timeline', label: t.requestTimeline, testId: 'detail-view-timeline' }, + { value: 'power', label: t.powerX, testId: 'detail-view-power' }, { value: 'aggregates', label: t.aggregatesAcrossConfigs, testId: 'detail-view-aggregates' }, { value: 'logs', label: t.logs, testId: 'detail-view-logs' }, ], @@ -347,6 +351,8 @@ export function AgenticPointDetail({ id }: Props) { {view === 'logs' ? ( + ) : view === 'power' ? ( + ) : view === 'aggregates' ? ( aggregatesQuery.isError ? ( { t: number; value: number }[]; +} + +const percent = (points: readonly { t: number; value: number }[]) => + points.map((p) => ({ t: p.t, value: p.value * 100 })); + +/** + * Menu of overlay candidates, in display order. Only sources whose series is + * non-empty for the point are offered (see `availableOverlaySources`). + */ +export const OVERLAY_SOURCES: readonly OverlaySource[] = [ + { + key: 'decodeTps', + label: { en: 'Decode throughput', zh: 'Decode 吞吐量' }, + unit: 'tok/s', + color: '#8b5cf6', + points: (m) => m.decodeTps, + }, + { + key: 'prefillTps', + label: { en: 'Prefill throughput', zh: 'Prefill 吞吐量' }, + unit: 'tok/s', + color: '#06b6d4', + points: (m) => m.prefillTps, + }, + { + key: 'kvCacheUsage', + label: { en: 'KV cache utilization', zh: 'KV cache 利用率' }, + unit: '%', + color: '#f59e0b', + points: (m) => percent(m.kvCacheUsage), + }, + { + key: 'hostKvCacheUsage', + label: { en: 'Host KV cache utilization', zh: '主机 KV cache 利用率' }, + unit: '%', + color: '#d97706', + points: (m) => percent(m.hostKvCacheUsage), + }, + { + key: 'prefixCacheHitRate', + label: { en: 'Prefix cache hit rate', zh: 'Prefix cache 命中率' }, + unit: '%', + color: '#10b981', + points: (m) => percent(m.prefixCacheHitRate), + }, + { + key: 'prefixCacheHitsTps', + label: { en: 'Prefix cache hits', zh: 'Prefix cache 命中量' }, + unit: 'tok/s', + color: '#14b8a6', + points: (m) => m.prefixCacheHitsTps, + }, + { + key: 'queueDepth', + label: { en: 'Queue depth (running + waiting)', zh: '队列深度(运行中 + 等待中)' }, + unit: 'req', + color: '#ec4899', + points: (m) => m.queueDepth.map((p) => ({ t: p.t, value: p.total })), + }, +]; + +/** Sources that have at least one sample for this point, in menu order. */ +export function availableOverlaySources( + metrics: TraceServerMetrics | null | undefined, +): OverlaySource[] { + if (!metrics) return []; + return OVERLAY_SOURCES.filter((source) => source.points(metrics).length > 0); +} + +export function overlaySourceLabel(source: OverlaySource, locale: Locale): string { + return source.label[locale]; +} diff --git a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx new file mode 100644 index 000000000..7d56ef9a2 --- /dev/null +++ b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx @@ -0,0 +1,448 @@ +'use client'; + +import { useMemo, useState } from 'react'; + +import GpuMetricsChart, { + GPU_COLORS, + type TelemetryOverlaySeries, +} from '@/components/gpu-power/GpuPowerChart'; +import GpuStatsTable from '@/components/gpu-power/GpuStatsTable'; +import { TelemetryDisplayControls } from '@/components/gpu-power/TelemetryDisplayControls'; +import { + DEFAULT_TELEMETRY_DISPLAY, + toAbsoluteMs, + type TelemetryDisplayState, +} from '@/components/gpu-power/telemetry-smoothing'; +import { + type GpuMetricKey, + type GpuMetricRow, + ALL_METRIC_OPTIONS, + getAvailableMetrics, + getGpuMetricLabel, +} from '@/components/gpu-power/types'; +import { Card } from '@/components/ui/card'; +import ChartLegend from '@/components/ui/chart-legend'; +import { Label } from '@/components/ui/label'; +import { RetryableQueryError } from '@/components/ui/retryable-query-error'; +import { + Select, + SelectContent, + SelectItem, + SelectTrigger, + SelectValue, +} from '@/components/ui/select'; +import { useGpuMetricsPoint, type GpuMetricSeries } from '@/hooks/api/use-gpu-metrics-point'; +import { useTraceServerMetrics } from '@/hooks/api/use-trace-server-metrics'; + +import { availableOverlaySources, overlaySourceLabel } from './overlay-sources'; +import { track } from '@/lib/analytics'; +import { useLocale } from '@/lib/use-locale'; + +const STRINGS = { + en: { + loading: 'Loading PowerX telemetry…', + error: 'Failed to load PowerX telemetry.', + missing: + 'No PowerX telemetry is stored for benchmark point #{id}. Telemetry may not have been collected or ingested.', + series: 'Telemetry series', + metric: 'Metric', + vendor: 'Collector', + samples: 'Samples', + chips: 'Chips', + interval: 'Sample interval', + window: 'Recorded window', + sharedNote: + 'This series covers the whole benchmark job, including server start-up and warm-up, so summary rows below span more than the measured serving window.', + perGpuStats: 'Per-chip statistics', + chip: 'Chip', + secondsUnit: 's', + resetFilter: 'Show all chips', + overlayToggle: 'Overlay server metric', + overlayNone: 'None', + overlayLoading: 'Loading server metrics…', + overlayError: 'Server metrics failed to load; overlays are unavailable.', + overlayUnavailable: 'This point has no server-metric series to overlay.', + overlayAligned: 'The overlay is aligned to the telemetry by wall-clock timestamps.', + overlayRelative: + 'The trace has no wall-clock timestamps, so the overlay and the telemetry are both aligned at their own t=0.', + }, + zh: { + loading: '正在加载 PowerX 遥测数据……', + error: 'PowerX 遥测数据加载失败。', + missing: '基准测试数据点 #{id} 没有存储的 PowerX 遥测数据。遥测数据可能尚未采集或入库。', + series: '遥测序列', + metric: '指标', + vendor: '采集器', + samples: '样本数', + chips: '芯片数', + interval: '采样间隔', + window: '记录时间窗口', + sharedNote: + '该序列覆盖整个基准测试任务,包括服务启动与 warmup 阶段,因此下方统计范围大于实际测量的服务窗口。', + perGpuStats: '单芯片统计信息', + chip: '芯片', + secondsUnit: '秒', + resetFilter: '显示全部芯片', + overlayToggle: '叠加服务端指标', + overlayNone: '无', + overlayLoading: '正在加载服务端指标……', + overlayError: '服务端指标加载失败,无法叠加显示。', + overlayUnavailable: '该数据点没有可叠加的服务端指标序列。', + overlayAligned: '叠加曲线已按绝对时间戳与遥测数据对齐。', + overlayRelative: 'trace 缺少绝对时间戳,因此叠加曲线与遥测数据均从各自的 t=0 开始对齐。', + }, +} as const; + +const VENDOR_LABEL: Record = { nvidia: 'nvidia-smi', amd: 'amd-smi' }; + +/** + * Single-node CSVs come from the vendor CLI; multinode power bundles record + * their own producer (e.g. `srt-slurm.dcgm-power`) in the context sidecar. + */ +export function collectorLabel(series: Pick): string { + const context = series.sidecars?.context; + const producer = + context && typeof context === 'object' ? (context as { producer?: unknown }).producer : null; + if (typeof producer === 'string' && producer.trim() !== '') return producer; + return VENDOR_LABEL[series.vendor] ?? series.vendor; +} + +interface Props { + id: number; + enabled: boolean; + /** The point's hardware key, for the TDP reference line. */ + hardware?: string; + /** Fixed-sequence points do not have AgentX server-metric overlays. */ + serverMetricsEnabled?: boolean; +} + +function seriesLabel(series: GpuMetricSeries, total: number): string { + return total > 1 ? `${series.artifactName} · ${series.fileName}` : series.artifactName; +} + +/** + * PowerX tab of the per-point detail page: the full-resolution chip telemetry + * recorded while this benchmark point ran, read from the ingest-time digest + * (migration 016) rather than from GitHub artifacts. + */ +export function PowerTelemetryView({ id, enabled, hardware, serverMetricsEnabled = true }: Props) { + const locale = useLocale(); + const t = STRINGS[locale]; + const query = useGpuMetricsPoint(id, enabled); + const seriesList = query.data?.series ?? []; + + const [seriesSelection, setSeriesSelection] = useState<{ id: number; seriesId: number } | null>( + null, + ); + const selectedSeries = + (seriesSelection?.id === id + ? seriesList.find((series) => series.id === seriesSelection.seriesId) + : undefined) ?? seriesList[0]; + const data: GpuMetricRow[] = useMemo(() => selectedSeries?.data ?? [], [selectedSeries]); + const availableMetrics = useMemo(() => getAvailableMetrics(data), [data]); + + const [metricSelection, setMetricSelection] = useState('power'); + const metricKey: GpuMetricKey = availableMetrics.some((m) => m.key === metricSelection) + ? metricSelection + : 'power'; + const metricConfig = ALL_METRIC_OPTIONS.find((m) => m.key === metricKey)!; + const allGpuIndices = useMemo( + () => [...new Set(data.map((row) => row.index))].toSorted((a, b) => a - b), + [data], + ); + // Hidden chips are scoped to the series they were hidden on so switching + // series never carries over a stale filter. + const [hiddenSelection, setHiddenSelection] = useState<{ + seriesId: number; + hidden: number[]; + } | null>(null); + const hiddenGpus = useMemo( + () => + new Set( + hiddenSelection && hiddenSelection.seriesId === selectedSeries?.id + ? hiddenSelection.hidden + : [], + ), + [hiddenSelection, selectedSeries?.id], + ); + const visibleGpus = useMemo( + () => new Set(allGpuIndices.filter((gpuIndex) => !hiddenGpus.has(gpuIndex))), + [allGpuIndices, hiddenGpus], + ); + const toggleGpu = (gpuIndex: number) => { + if (!selectedSeries) return; + track('inference_agentic_power_gpu_toggled', { id, gpuIndex }); + const next = new Set(hiddenGpus); + if (next.has(gpuIndex)) next.delete(gpuIndex); + else next.add(gpuIndex); + setHiddenSelection({ seriesId: selectedSeries.id, hidden: [...next] }); + }; + const [isLegendExpanded, setIsLegendExpanded] = useState(true); + const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); + + // Server-metric overlay. The series are fetched as soon as the tab opens so + // the menu can list exactly the metrics this point has; one source at a time. + const metricsQuery = useTraceServerMetrics(id, enabled && serverMetricsEnabled); + const serverMetrics = metricsQuery.data; + const overlaySources = useMemo(() => availableOverlaySources(serverMetrics), [serverMetrics]); + const [overlaySelection, setOverlaySelection] = useState<{ id: number; key: string } | null>( + null, + ); + const overlayKey = overlaySelection?.id === id ? overlaySelection.key : 'none'; + const overlaySource = overlaySources.find((source) => source.key === overlayKey) ?? null; + // Trace timeslices carry epoch-ns starts, so both series can share wall-clock + // time. A zero startNs means the trace only has relative time. + const overlayAbsolute = Boolean(serverMetrics && serverMetrics.startNs > 0); + const overlay = useMemo(() => { + if (!overlaySource || !serverMetrics || !selectedSeries) return null; + const originMs = overlayAbsolute + ? serverMetrics.startNs / 1e6 + : new Date(selectedSeries.startedAt).getTime(); + return { + label: overlaySourceLabel(overlaySource, locale), + unit: overlaySource.unit, + color: overlaySource.color, + points: toAbsoluteMs(overlaySource.points(serverMetrics), originMs), + }; + }, [overlaySource, serverMetrics, selectedSeries, overlayAbsolute, locale]); + const overlayNote = ((): string | null => { + if (metricsQuery.isLoading) return t.overlayLoading; + if (metricsQuery.isError) return t.overlayError; + if (overlaySources.length === 0) return t.overlayUnavailable; + if (!overlay) return null; + return overlayAbsolute ? t.overlayAligned : t.overlayRelative; + })(); + + if (!enabled) return null; + + if (query.isLoading) { + return ( +
+ {t.loading} +
+ ); + } + if (query.isError && !query.data) { + return ( + + ); + } + if (!selectedSeries) { + return ( +
+ {t.missing.replace('{id}', String(id))} +
+ ); + } + + const durationS = Math.max( + 0, + (new Date(selectedSeries.endedAt).getTime() - new Date(selectedSeries.startedAt).getTime()) / + 1000, + ); + const numberLocale = locale === 'zh' ? 'zh-CN' : undefined; + + return ( +
+ +
+
+
{t.vendor}
+
{collectorLabel(selectedSeries)}
+
+
+
{t.samples}
+
+ {selectedSeries.sampleCount.toLocaleString(numberLocale)} +
+
+
+
{t.chips}
+
{selectedSeries.gpuCount}
+
+
+
{t.interval}
+
+ {selectedSeries.sampleIntervalS === null + ? '—' + : `${selectedSeries.sampleIntervalS.toFixed(2)} ${t.secondsUnit}`} +
+
+
+
{t.window}
+
+ {new Date(selectedSeries.startedAt).toLocaleTimeString(numberLocale)} ·{' '} + {Math.round(durationS).toLocaleString(numberLocale)} {t.secondsUnit} +
+
+
+
+ {seriesList.length > 1 && ( +
+ + +
+ )} +
+ + +
+
+ + {serverMetricsEnabled && ( +
+
+ + +
+ {overlayNote && ( + + {overlayNote} + + )} +
+ )} +
+ + + ({ + name: `${t.chip} ${gpuIndex}`, + hw: String(gpuIndex), + label: `${t.chip} ${gpuIndex}`, + color: GPU_COLORS[gpuIndex % GPU_COLORS.length], + isActive: visibleGpus.has(gpuIndex), + onClick: () => toggleGpu(gpuIndex), + }))} + onItemRemove={(hw) => { + const gpuIndex = Number(hw); + if (visibleGpus.has(gpuIndex)) toggleGpu(gpuIndex); + }} + isLegendExpanded={isLegendExpanded} + onExpandedChange={(expanded) => { + setIsLegendExpanded(expanded); + track('inference_agentic_power_legend_expanded', { id, expanded }); + }} + actions={ + hiddenGpus.size === 0 + ? [] + : [ + { + id: 'power-telemetry-show-all-chips', + label: t.resetFilter, + onClick: () => { + track('inference_agentic_power_gpu_reset_filter', { id }); + setHiddenSelection(null); + }, + }, + ] + } + /> + } + caption={ + + {getGpuMetricLabel(metricConfig, locale)} · {t.sharedNote} + + } + /> + + + +

{t.perGpuStats}

+ +
+
+ ); +} diff --git a/packages/app/src/components/inference/agentic-point/use-detail-view.ts b/packages/app/src/components/inference/agentic-point/use-detail-view.ts index 1a97a85cc..b08b3a868 100644 --- a/packages/app/src/components/inference/agentic-point/use-detail-view.ts +++ b/packages/app/src/components/inference/agentic-point/use-detail-view.ts @@ -6,10 +6,14 @@ import { useClientSearchParams } from '@/hooks/useClientSearch'; import { track } from '@/lib/analytics'; import { replaceClientSearch } from '@/lib/client-navigation'; -export type DetailView = 'point' | 'timeline' | 'aggregates' | 'logs'; +export type DetailView = 'point' | 'timeline' | 'power' | 'aggregates' | 'logs'; const isDetailView = (value: string | null): value is DetailView => - value === 'point' || value === 'timeline' || value === 'aggregates' || value === 'logs'; + value === 'point' || + value === 'timeline' || + value === 'power' || + value === 'aggregates' || + value === 'logs'; /** URL-persisted detail view (`?view=`; per-point is the unadorned default). */ export function useDetailView(): [DetailView, (nextView: DetailView) => void] { diff --git a/packages/app/src/components/inference/axis-metric-explanations.ts b/packages/app/src/components/inference/axis-metric-explanations.ts index 220703207..5ce643e70 100644 --- a/packages/app/src/components/inference/axis-metric-explanations.ts +++ b/packages/app/src/components/inference/axis-metric-explanations.ts @@ -431,6 +431,116 @@ export const METRIC_EXPLANATIONS: Record = { zh: '% TDP = 每芯片实测平均功耗(W)÷ 额定 TDP(W)× 100', }, }, + measuredPowerTimeline: { + description: { + en: + `The per-second accelerator power samples behind each measured average, drawn over ` + + `the whole benchmark job (server start, warmup, and the validated measurement window, ` + + `which is emphasized). One trace per config, mean of its GPUs by default; the rated TDP ` + + `is a dashed reference per hardware. Configs whose telemetry artifact is missing are ` + + `listed under the chart rather than estimated.${MEASURED_TIER_NOTE_EN}`, + zh: + `每个实测平均值背后的逐秒加速器功耗采样,覆盖整个基准测试任务(服务启动、warmup ` + + `以及被突出显示的有效测量窗口)。每个配置一条曲线,默认取其 GPU 的平均值;` + + `每种硬件的额定 TDP 以虚线作为参考。缺少遥测产物的配置会列在图表下方,而不会用估算值代替。${ + MEASURED_TIER_NOTE_ZH + }`, + }, + formula: { + en: 'W(t) = mean over GPUs of the sampled power draw in each one-second bucket', + zh: 'W(t) = 每个一秒时间桶内各 GPU 功耗采样值的平均', + }, + }, + gpuProvisionedWatts: { + description: { + en: + 'Rated accelerator TDP from the hardware registry, shown as a flat per-chip value so ' + + 'measured power can be read against the GPU-only provisioning boundary. It does not ' + + 'depend on the run.', + zh: + '取硬件注册表中的加速器额定 TDP,以每芯片恒定值显示,用于对照 GPU 侧的额定供电边界与实测功耗。' + + '该值与具体运行无关。', + }, + formula: { + en: 'W/GPU = rated TDP (W)', + zh: 'W/GPU = 额定 TDP(W)', + }, + }, + gpuProvisionedJPerOutputToken: { + description: { + en: + 'Energy per output token if every allocated accelerator drew exactly its rated TDP for ' + + 'the whole run. Disaggregated deployments count prefill and decode GPUs together, so ' + + 'this is the GPU-only provisioning boundary the measured J/token can be compared against.', + zh: + '假设所有已分配加速器在整个运行中恒以额定 TDP 耗电时的每输出 token 能耗。' + + '分离式部署将 prefill 与 decode GPU 一并计入,因此它是可与实测 J/token 对照的 GPU 侧额定边界。', + }, + formula: { + en: 'J/tok = rated TDP (W) × allocated GPUs ÷ total output tokens per second', + zh: 'J/tok = 额定 TDP(W)× 已分配 GPU 数 ÷ 总输出 token 吞吐(tok/s)', + }, + }, + utilityProvisionedWatts: { + description: { + en: + 'All-in provisioned power per chip from the hardware registry: the utility-side capacity ' + + 'a data center reserves for one accelerator including host, networking, cooling and ' + + 'power-conversion overheads. It is a flat value independent of the run.', + zh: + '取硬件注册表中的每芯片全电源配置功耗:数据中心为单张加速器预留的电源侧容量,' + + '包含主机、网络、散热与电源转换开销。该值为恒定值,与运行无关。', + }, + formula: { + en: 'W/GPU = all-in provisioned power per GPU (kW) × 1000', + zh: 'W/GPU = 每 GPU 全电源配置功耗(kW)× 1000', + }, + }, + utilityProvisionedJPerOutputToken: { + description: { + en: + 'Energy per output token at the all-in provisioned power boundary, normalized by every ' + + 'allocated accelerator. It differs from the public All-in Provisioned J per Output Token ' + + 'metric only for disaggregated runs, where that metric normalizes by decode GPUs alone.', + zh: + '在全电源配置边界下的每输出 token 能耗,按全部已分配加速器归一。' + + '仅在分离式运行中与公开的 All-in Provisioned J per Output Token 指标不同,后者只按 decode GPU 归一。', + }, + formula: { + en: 'J/tok = all-in provisioned power per GPU (W) × allocated GPUs ÷ total output tokens per second', + zh: 'J/tok = 每 GPU 全电源配置功耗(W)× 已分配 GPU 数 ÷ 总输出 token 吞吐(tok/s)', + }, + }, + utilityModeledWatts: { + description: { + en: + 'Modeled facility power per allocated accelerator: measured GPU power is scaled to chassis ' + + 'AC by the system power model and then multiplied once by PUE. Only hardware with a known ' + + 'eight-GPU chassis profile on 8k1k runs is supported; NVL72 systems show no value.', + zh: + '每已分配加速器的数据中心建模功耗:先由系统功耗模型将 GPU 实测功耗换算为机箱交流功耗,再乘以一次 PUE。' + + '仅支持在 8k1k 运行中具有已知八卡机箱模型的硬件;NVL72 系统不显示数值。', + }, + formula: { + en: 'W/GPU = modeled chassis AC power (W) × PUE ÷ allocated GPUs', + zh: 'W/GPU = 机箱交流建模功耗(W)× PUE ÷ 已分配 GPU 数', + }, + }, + utilityModeledJPerOutputToken: { + description: { + en: + 'Measured energy per output token scaled to the modeled facility boundary, so its ratio to ' + + 'measured GPU energy equals the ratio of modeled facility power to measured GPU power. ' + + 'Missing where the system power model or validated measured power is unavailable.', + zh: + '将实测每输出 token 能耗按建模的数据中心边界缩放,其与 GPU 实测能耗之比等于数据中心建模功耗与 GPU 实测功耗之比。' + + '系统功耗模型或通过验证的实测功耗缺失时不显示。', + }, + formula: { + en: 'J/tok = measured J per output token × modeled facility W per GPU ÷ measured W per GPU', + zh: 'J/tok = 实测每输出 token 能耗 × 每 GPU 数据中心建模功耗(W)÷ 每 GPU 实测功耗(W)', + }, + }, }; /** diff --git a/packages/app/src/components/inference/hooks/useChartData.ts b/packages/app/src/components/inference/hooks/useChartData.ts index 9f6498b5a..791b14be6 100644 --- a/packages/app/src/components/inference/hooks/useChartData.ts +++ b/packages/app/src/components/inference/hooks/useChartData.ts @@ -27,12 +27,14 @@ import type { ChartDefinition, HardwareConfig, InferenceData, + PowerCompare, RenderableGraph, TokenRevenuePriceSource, TokenRevenuePricing, YAxisMetricKey, } from '@/components/inference/types'; import { partitionChartDataByLimits } from '@/components/inference/utils'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; import { parseComparisonEntry } from '@/components/inference/utils/comparisonEntry'; import { computeAvailableQuickFilters, @@ -116,6 +118,8 @@ export function useChartData( tcoBasis: TcoBasis = DEFAULT_TCO_BASIS, /** Opt-in from the inference page only; explicit date/run/history views opt out. */ allowDefaultRunPreference = false, + /** Sibling boundary / role series appended to a gated power metric (`i_pcompare`). */ + powerCompare: PowerCompare = 'none', ) { // When the selected date is the latest available, use '' (empty string) to match // the initial no-date query key, reusing the eagerly-fetched benchmarks from the @@ -538,8 +542,14 @@ export function useChartData( ); const hasMetric = metricData.length > 0; const isTtftX = typeof xAxisField === 'string' && xAxisField.endsWith('_ttft'); + // Comparison clones are appended after the remap so they share the + // base point's x and differ only in y and `powerVariant`. const mappedData = hasMetric - ? metricData.map((d) => remapInferencePoint(d, metricKey, xAxisField)) + ? expandPowerCompareSeries( + metricData.map((d) => remapInferencePoint(d, metricKey, xAxisField)), + selectedYAxisMetric, + powerCompare, + ) : []; const isAgentic = selectedSequence === Sequence.AgenticTraces; @@ -576,6 +586,7 @@ export function useChartData( compareGpuPair, selectedPercentile, quickFilters, + powerCompare, ]); // Points that pass every scope filter but NOT the y-metric coverage filter. diff --git a/packages/app/src/components/inference/measured-metric-config.test.ts b/packages/app/src/components/inference/measured-metric-config.test.ts index 47c356f22..41276d400 100644 --- a/packages/app/src/components/inference/measured-metric-config.test.ts +++ b/packages/app/src/components/inference/measured-metric-config.test.ts @@ -1,6 +1,10 @@ import { describe, expect, it } from 'vitest'; -import { MEASURED_ENERGY_METRIC_CONFIG_KEYS, METRIC_CONFIG_KEYS } from './metric-registry'; +import { + MEASURED_ENERGY_METRIC_CONFIG_KEYS, + METRIC_CONFIG_KEYS, + POWER_BASIS_METRIC_CONFIG_KEYS, +} from './metric-registry'; import { changeMeasuredMetricConfig, getMeasuredMetricConfig, @@ -8,20 +12,27 @@ import { } from './measured-metric-config'; describe('measured metric configuration', () => { - it.each(MEASURED_ENERGY_METRIC_CONFIG_KEYS)( - 'round-trips the existing share-link metric %s', - (key) => { - const config = getMeasuredMetricConfig(key); - expect(config).toBeDefined(); - expect(changeMeasuredMetricConfig(key, {})).toBe(key); - expect(changeMeasuredMetricConfig('y_tpPerGpu', config!)).toBe(key); - }, - ); + it.each([ + 'y_measuredAvgPower', + 'y_measuredPowerTimeline', + 'y_measuredJPerOutputToken', + 'y_measuredWhPerSuccessfulQuery', + ] as const)('round-trips the existing share-link metric %s', (key) => { + const config = getMeasuredMetricConfig(key); + expect(config).toBeDefined(); + expect(changeMeasuredMetricConfig(key, {})).toBe(key); + expect(changeMeasuredMetricConfig('y_tpPerGpu', config!)).toBe(key); + }); it('does not group unrelated metrics or unknown persisted values', () => { const grouped = METRIC_CONFIG_KEYS.filter((key) => getMeasuredMetricConfig(key)); - expect(grouped).toHaveLength(13); - expect(new Set(grouped)).toEqual(new Set(MEASURED_ENERGY_METRIC_CONFIG_KEYS)); + expect(grouped).toHaveLength(20); + expect(new Set(grouped)).toEqual( + new Set([...MEASURED_ENERGY_METRIC_CONFIG_KEYS, ...POWER_BASIS_METRIC_CONFIG_KEYS]), + ); + for (const key of MEASURED_ENERGY_METRIC_CONFIG_KEYS) { + expect(getMeasuredMetricConfig(key)?.basis, key).toBe('gpu-measured'); + } expect(getMeasuredMetricConfig('y_modeledChassisPowerPerGpu')).toBeUndefined(); expect(getMeasuredMetricConfig('y_removedMetric')).toBeUndefined(); expect(getMeasuredMetricConfig('')).toBeUndefined(); @@ -42,6 +53,7 @@ describe('measured metric configuration', () => { it('keeps fleet percentiles, role averages and TDP normalization distinct', () => { expect(getMeasuredMetricConfig('y_measuredP90Power')).toEqual({ family: 'power', + basis: 'gpu-measured', scope: 'all', statistic: 'p90', display: 'watts', @@ -78,6 +90,7 @@ describe('measured metric configuration', () => { ); expect(getMeasuredMetricConfig('y_measuredPrefillJPerInputToken')).toEqual({ family: 'energy', + basis: 'gpu-measured', scope: 'prefill', denominator: 'input', unit: 'joules', diff --git a/packages/app/src/components/inference/measured-metric-config.ts b/packages/app/src/components/inference/measured-metric-config.ts index 555fbef2e..ff28546be 100644 --- a/packages/app/src/components/inference/measured-metric-config.ts +++ b/packages/app/src/components/inference/measured-metric-config.ts @@ -1,17 +1,27 @@ +import type { PowerBasis } from '@/lib/power-basis'; import type { MetricConfigKey } from './metric-registry'; export type MeasuredMetricFamily = 'power' | 'energy'; type MeasuredScope = 'all' | 'prefill' | 'decode'; +/** + * How whole-deployment average power is shown: per-chip watts, percent of + * TDP, or the per-second telemetry trace behind the average (`timeline`, which + * ChartDisplay renders with `PowerTimeline` instead of the scatter chart). + */ +export type MeasuredPowerDisplay = 'watts' | 'tdp' | 'timeline'; export type MeasuredMetricConfig = | { family: 'power'; + /** Power boundary the key plots; only `gpu-measured` publishes the other dimensions. */ + basis: PowerBasis; scope: MeasuredScope; statistic: 'average' | 'p75' | 'p90'; - display: 'watts' | 'tdp'; + display: MeasuredPowerDisplay; } | { family: 'energy'; + basis: PowerBasis; scope: MeasuredScope; denominator: 'input' | 'output' | 'total' | 'query'; unit: 'joules' | 'wattHours'; @@ -19,9 +29,10 @@ export type MeasuredMetricConfig = export type MeasuredMetricConfigChange = Partial<{ family: MeasuredMetricFamily; + basis: PowerBasis; scope: MeasuredScope; statistic: 'average' | 'p75' | 'p90'; - display: 'watts' | 'tdp'; + display: MeasuredPowerDisplay; denominator: 'input' | 'output' | 'total' | 'query'; unit: 'joules' | 'wattHours'; }>; @@ -31,51 +42,78 @@ export const MEASURED_METRIC_DEFAULTS = { energy: 'y_measuredJPerOutputToken', } as const satisfies Record; +const measured = { basis: 'gpu-measured' } as const; + // Presentation settings resolve to existing metrics; they do not own chart state. const MEASURED_METRIC_CONFIGS: readonly (readonly [MetricConfigKey, MeasuredMetricConfig])[] = [ - ['y_measuredAvgPower', { family: 'power', scope: 'all', statistic: 'average', display: 'watts' }], - ['y_measuredP75Power', { family: 'power', scope: 'all', statistic: 'p75', display: 'watts' }], - ['y_measuredP90Power', { family: 'power', scope: 'all', statistic: 'p90', display: 'watts' }], + [ + 'y_measuredAvgPower', + { family: 'power', ...measured, scope: 'all', statistic: 'average', display: 'watts' }, + ], + [ + 'y_measuredP75Power', + { family: 'power', ...measured, scope: 'all', statistic: 'p75', display: 'watts' }, + ], + [ + 'y_measuredP90Power', + { family: 'power', ...measured, scope: 'all', statistic: 'p90', display: 'watts' }, + ], [ 'y_measuredPrefillAvgPower', - { family: 'power', scope: 'prefill', statistic: 'average', display: 'watts' }, + { family: 'power', ...measured, scope: 'prefill', statistic: 'average', display: 'watts' }, ], [ 'y_measuredDecodeAvgPower', - { family: 'power', scope: 'decode', statistic: 'average', display: 'watts' }, + { family: 'power', ...measured, scope: 'decode', statistic: 'average', display: 'watts' }, ], [ 'y_measuredPowerPercentTdp', - { family: 'power', scope: 'all', statistic: 'average', display: 'tdp' }, + { family: 'power', ...measured, scope: 'all', statistic: 'average', display: 'tdp' }, + ], + [ + 'y_measuredPowerTimeline', + { family: 'power', ...measured, scope: 'all', statistic: 'average', display: 'timeline' }, ], [ 'y_measuredJPerInputToken', - { family: 'energy', scope: 'all', denominator: 'input', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'input', unit: 'joules' }, ], [ 'y_measuredJPerOutputToken', - { family: 'energy', scope: 'all', denominator: 'output', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'output', unit: 'joules' }, ], [ 'y_measuredJPerTotalToken', - { family: 'energy', scope: 'all', denominator: 'total', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'total', unit: 'joules' }, ], [ 'y_measuredPrefillJPerInputToken', - { family: 'energy', scope: 'prefill', denominator: 'input', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'prefill', denominator: 'input', unit: 'joules' }, ], [ 'y_measuredDecodeJPerOutputToken', - { family: 'energy', scope: 'decode', denominator: 'output', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'decode', denominator: 'output', unit: 'joules' }, ], [ 'y_measuredJPerSuccessfulQuery', - { family: 'energy', scope: 'all', denominator: 'query', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'query', unit: 'joules' }, ], [ 'y_measuredWhPerSuccessfulQuery', - { family: 'energy', scope: 'all', denominator: 'query', unit: 'wattHours' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'query', unit: 'wattHours' }, ], + // Derived boundaries publish one canonical combination per family: whole + // deployment, average watts, joules per output token (lib/power-basis.ts). + ...( + [ + ['gpu-provisioned', 'y_gpuProvisionedWatts', 'y_gpuProvisionedJPerOutputToken'], + ['utility-provisioned', 'y_utilityProvisionedWatts', 'y_utilityProvisionedJPerOutputToken'], + ['utility-modeled', 'y_utilityModeledWatts', 'y_utilityModeledJPerOutputToken'], + ] as const satisfies readonly (readonly [PowerBasis, MetricConfigKey, MetricConfigKey])[] + ).flatMap(([basis, watts, energy]): (readonly [MetricConfigKey, MeasuredMetricConfig])[] => [ + [watts, { family: 'power', basis, scope: 'all', statistic: 'average', display: 'watts' }], + [energy, { family: 'energy', basis, scope: 'all', denominator: 'output', unit: 'joules' }], + ]), ]; export function getMeasuredMetricConfig(metric: string): MeasuredMetricConfig | undefined { @@ -83,6 +121,8 @@ export function getMeasuredMetricConfig(metric: string): MeasuredMetricConfig | return config ? { ...config } : undefined; } +const OTHER_DIMENSIONS = ['scope', 'statistic', 'display', 'denominator', 'unit'] as const; + export function changeMeasuredMetricConfig( metric: string, change: MeasuredMetricConfigChange, @@ -93,6 +133,21 @@ export function changeMeasuredMetricConfig( current?.family === family ? current : getMeasuredMetricConfig(MEASURED_METRIC_DEFAULTS[family])!; + // Choosing a derived boundary snaps the other dimensions to its canonical + // combination. Changing any of those dimensions while on a derived boundary + // returns to GPU-measured telemetry, the only basis that publishes variants, + // so every control change lands on a real key. A family switch keeps the + // boundary: the metric key is what carries it. + const changesOtherDimension = OTHER_DIMENSIONS.some((key) => change[key] !== undefined); + const basis = + change.basis ?? (changesOtherDimension ? 'gpu-measured' : (current ?? config).basis); + if (basis !== 'gpu-measured') { + return ( + MEASURED_METRIC_CONFIGS.find( + ([, candidate]) => candidate.family === family && candidate.basis === basis, + )?.[0] ?? MEASURED_METRIC_DEFAULTS[family] + ); + } let scope = change.scope ?? config.scope; if (config.family === 'power') { @@ -103,6 +158,7 @@ export function changeMeasuredMetricConfig( MEASURED_METRIC_CONFIGS.find( ([, candidate]) => candidate.family === 'power' && + candidate.basis === 'gpu-measured' && candidate.scope === scope && candidate.statistic === statistic && candidate.display === display, @@ -123,6 +179,7 @@ export function changeMeasuredMetricConfig( MEASURED_METRIC_CONFIGS.find( ([, candidate]) => candidate.family === 'energy' && + candidate.basis === 'gpu-measured' && candidate.scope === scope && candidate.denominator === denominator && candidate.unit === unit, diff --git a/packages/app/src/components/inference/metric-registry.test.ts b/packages/app/src/components/inference/metric-registry.test.ts index 28045728e..db4011826 100644 --- a/packages/app/src/components/inference/metric-registry.test.ts +++ b/packages/app/src/components/inference/metric-registry.test.ts @@ -19,6 +19,7 @@ import { metricCostTier, metricForCostTier, metricOptionTitle, + POWER_BASIS_METRIC_CONFIG_KEYS, resolveMetricConfigKey, tokenMetricTypeForConfigKey, } from './metric-registry'; @@ -39,6 +40,10 @@ describe('metric registry', () => { expect(e2e.y_costh_roofline).toBe('lower_left'); expect(interactivity.y_measuredPowerPercentTdp_roofline).toBe('lower_right'); expect(e2e.y_measuredPowerPercentTdp_roofline).toBe('lower_left'); + for (const key of POWER_BASIS_METRIC_CONFIG_KEYS) { + expect(interactivity[`${key}_roofline`], key).toBe('lower_right'); + expect(e2e[`${key}_roofline`], key).toBe('lower_left'); + } }); it('preserves metric-specific x overrides and bilingual labels', () => { @@ -210,7 +215,11 @@ describe('metric registry', () => { ); const measuredGroup = METRIC_CONTROL_GROUPS.find((group) => group.label === 'Measured Energy'); - expect(measuredGroup?.metrics).toBe(MEASURED_ENERGY_METRIC_CONFIG_KEYS); + expect(measuredGroup?.gated).toBe(true); + expect(measuredGroup?.metrics).toEqual([ + ...MEASURED_ENERGY_METRIC_CONFIG_KEYS, + ...POWER_BASIS_METRIC_CONFIG_KEYS, + ]); }); it('classifies measured-energy config keys', () => { diff --git a/packages/app/src/components/inference/metric-registry.ts b/packages/app/src/components/inference/metric-registry.ts index ccba918b1..9687c561d 100644 --- a/packages/app/src/components/inference/metric-registry.ts +++ b/packages/app/src/components/inference/metric-registry.ts @@ -386,6 +386,80 @@ export const METRIC_REGISTRY = { titleZh: '实测平均功耗占 TDP 百分比', polarity: 'lower', }, + // The per-second telemetry behind `measuredAvgPower`. The field aliases the + // same average so the table view, availability panel, and share links keep + // working; ChartDisplay swaps the scatter chart for `PowerTimeline`, which + // fetches each point's `gpu_metrics_*` artifact and draws the trace. The + // label leads with "Measured Average Power", like the %TDP display, so a + // "Measured Power" search still finds only the family option. + measuredPowerTimeline: { + field: 'measuredPowerTimeline.y', + label: 'Measured Average Power per Chip over Time (W)', + labelZh: '每芯片实测平均功耗时间线(W)', + title: 'Measured Average Power per Chip over Time', + titleZh: '每芯片实测平均功耗时间线', + polarity: 'lower', + }, + // Power boundaries beyond GPU-measured telemetry (`lib/power-basis.ts`). + // Each boundary publishes W per allocated GPU and J per output token; the + // Boundary select in the Measured controls resolves to these keys, so the + // metric key alone carries the boundary in share links. Keys deliberately + // lack the `measured` prefix: they are spec constants or model output, not + // telemetry, so the telemetry-only decorations must not treat them as such. + gpuProvisionedWatts: { + field: 'gpuProvisionedWatts.y', + label: 'GPU Provisioned Power per Chip (TDP, W)', + labelZh: '每芯片 GPU 额定功耗(TDP,W)', + title: 'GPU Provisioned Power per Chip (TDP)', + titleZh: '每芯片 GPU 额定功耗(TDP)', + polarity: 'lower', + }, + gpuProvisionedJPerOutputToken: { + field: 'gpuProvisionedJPerOutputToken.y', + label: 'GPU Provisioned J per Output Token (TDP, J/tok)', + labelZh: '每输出 token GPU 额定能耗(TDP,J/tok)', + title: 'GPU Provisioned Joules per Output Token (TDP)', + titleZh: '每输出 token GPU 额定焦耳能耗(TDP)', + polarity: 'lower', + }, + utilityProvisionedWatts: { + field: 'utilityProvisionedWatts.y', + label: 'Utility Provisioned Power per Chip (all-in, W)', + labelZh: '每芯片全电源配置功耗(all-in,W)', + title: 'Utility Provisioned Power per Chip (all-in)', + titleZh: '每芯片全电源配置功耗(all-in)', + polarity: 'lower', + }, + // Unlike the ungated `jOutput`, which divides by output tokens per decode + // GPU, this normalizes by every allocated GPU (prefill + decode). + utilityProvisionedJPerOutputToken: { + field: 'utilityProvisionedJPerOutputToken.y', + label: 'Utility Provisioned J per Output Token, all GPUs (all-in, J/tok)', + labelZh: '每输出 token 全电源配置能耗,按全部 GPU 归一(all-in,J/tok)', + title: 'Utility Provisioned Joules per Output Token, all GPUs (all-in)', + titleZh: '每输出 token 全电源配置焦耳能耗,按全部 GPU 归一(all-in)', + polarity: 'lower', + }, + // zh vocabulary shared with the Boundary select, its help text and the chart + // caption: B3 “全电源配置” (as the ungated jOutput/jTotal already say for + // all-in), B4 “数据中心建模” (measured GPU power carried through the chassis + // model to the utility meter). + utilityModeledWatts: { + field: 'utilityModeledWatts.y', + label: 'Utility Modeled Power per Chip (PUE, W)', + labelZh: '每芯片数据中心建模功耗(含 PUE,W)', + title: 'Utility Modeled Power per Chip (PUE)', + titleZh: '每芯片数据中心建模功耗(含 PUE)', + polarity: 'lower', + }, + utilityModeledJPerOutputToken: { + field: 'utilityModeledJPerOutputToken.y', + label: 'Utility Modeled J per Output Token (PUE, J/tok)', + labelZh: '每输出 token 数据中心建模能耗(含 PUE,J/tok)', + title: 'Utility Modeled Joules per Output Token (PUE)', + titleZh: '每输出 token 数据中心建模焦耳能耗(含 PUE)', + polarity: 'lower', + }, } as const satisfies Record; export type MetricKey = keyof typeof METRIC_REGISTRY; @@ -597,6 +671,7 @@ export const MEASURED_ENERGY_METRIC_CONFIG_KEYS = [ 'y_measuredJPerSuccessfulQuery', 'y_measuredWhPerSuccessfulQuery', 'y_measuredPowerPercentTdp', + 'y_measuredPowerTimeline', ] as const satisfies readonly MetricConfigKey[]; const MEASURED_ENERGY_METRIC_CONFIG_KEY_SET: ReadonlySet = new Set( @@ -618,6 +693,31 @@ export function isRoleLocalMeasuredEnergyConfigKey(configKey: string): boolean { return ROLE_LOCAL_MEASURED_ENERGY_METRIC_CONFIG_KEY_SET.has(configKey); } +/** + * The derived power-boundary y-axes (GPU provisioned, utility provisioned, + * utility modeled) that share the gated Measured Energy group and its + * Boundary select. They are kept out of `MEASURED_ENERGY_METRIC_CONFIG_KEYS` + * on purpose: spec constants and model output carry no telemetry tier, so the + * legacy-power ring, tier tooltip line, and footer key do not apply to them. + */ +export const POWER_BASIS_METRIC_CONFIG_KEYS = [ + 'y_gpuProvisionedWatts', + 'y_gpuProvisionedJPerOutputToken', + 'y_utilityProvisionedWatts', + 'y_utilityProvisionedJPerOutputToken', + 'y_utilityModeledWatts', + 'y_utilityModeledJPerOutputToken', +] as const satisfies readonly MetricConfigKey[]; + +const POWER_BASIS_METRIC_CONFIG_KEY_SET: ReadonlySet = new Set( + POWER_BASIS_METRIC_CONFIG_KEYS, +); + +/** Whether a y-axis config key plots a derived power boundary (B2–B4). */ +export function isPowerBasisConfigKey(configKey: string): boolean { + return POWER_BASIS_METRIC_CONFIG_KEY_SET.has(configKey); +} + export const MODELED_SYSTEM_POWER_METRIC_CONFIG_KEY = 'y_modeledChassisPowerPerGpu'; /** Whether a y-axis config key plots the modeled chassis AC power metric. */ @@ -671,10 +771,13 @@ export const METRIC_CONTROL_GROUPS: readonly MetricControlGroup[] = [ // Runner power telemetry and the chassis model built on it are still being // validated, so both groups stay behind the ↑↑↓↓ feature gate until the // measurements are stable enough to publish. + // The derived boundaries ride along so the same gate and the same + // shared-URL exception (a gated metric selected by `i_metric` still renders + // while locked) apply to them. { label: 'Measured Energy', labelZh: '实测能耗', - metrics: MEASURED_ENERGY_METRIC_CONFIG_KEYS, + metrics: [...MEASURED_ENERGY_METRIC_CONFIG_KEYS, ...POWER_BASIS_METRIC_CONFIG_KEYS], gated: true, }, { diff --git a/packages/app/src/components/inference/perf-ruler-store.ts b/packages/app/src/components/inference/perf-ruler-store.ts new file mode 100644 index 000000000..98bbfd692 --- /dev/null +++ b/packages/app/src/components/inference/perf-ruler-store.ts @@ -0,0 +1,179 @@ +'use client'; + +import { + type Dispatch, + type SetStateAction, + createContext, + useCallback, + useContext, + useMemo, + useRef, + useState, +} from 'react'; + +import { track } from '@/lib/analytics'; +import { perfRulerAxisMetricKey } from '@/hooks/usePerfRulerAxisReset'; +import { + EMPTY_PERF_RULER_STATE, + MAX_PERF_RULERS, + type PerfRulerMeasurement, + type PerfRulerState, + clearPerfRulers, + parsePerfRulers, +} from '@/lib/d3-chart/layers/perf-ruler'; + +/** + * @file perf-ruler-store.ts + * @description Provider-owned Perf Ruler state for the primary inference + * chart, so completed rulers persist in share links (`i_rulers`). Lives + * beside `InferenceContext` rather than inside it so `ScatterGraph` can + * consume the store without importing the (heavily mocked) provider module. + */ + +/** + * The chart instance whose Perf Rulers persist in share links. `ChartDisplay` + * mounts the primary chart as `chart-${graphIndex}` and only graph 0 is ever + * visible; the replay chart (`replay-chart-0`) draws interpolated frames of + * the same curves and must NOT bind, or every ruler would render twice and + * the replay's prune pass could delete rulers the main chart still shows. + */ +export const PERSISTED_PERF_RULER_CHART_ID = 'chart-0'; + +/** + * Perf-ruler store for the persisted chart. Lives in its own context rather + * than the Display domain so a ruler commit does not rerender every display + * consumer, and so harnesses that mount `InferenceContextsProvider` with + * static mock values (no store) keep the chart's component-local fallback. + * + * `state` holds COMMITTED rulers: the D3 layer renders them and `i_rulers` + * serializes them. `pending` holds rulers parsed from the share link whose + * curves may not have been drawn yet — data, `i_gpus`, comparison dates, + * and `?unofficialrun=` overlays all arrive after the chart's first draw, + * and the chart prunes any committed ruler whose curve path is absent from + * the DOM. The chart therefore commits a pending ruler only once BOTH of + * its curve paths exist (see the perf-ruler decoration effect in + * ScatterGraph); rulers whose curves never appear stay pending, invisible + * and unserialized, until an axis change or an explicit clear discards them. + */ +export interface PerfRulerStore { + chartId: string; + state: PerfRulerState; + setState: Dispatch>; + pending: readonly PerfRulerMeasurement[] | null; + /** + * Commit share-link rulers whose curves now exist (`resolved`, iso-x + * already clamped to the pair's overlap) and keep `remaining` pending. + */ + commitPending: ( + resolved: readonly PerfRulerMeasurement[], + remaining: readonly PerfRulerMeasurement[] | null, + ) => void; + /** Drop share-link rulers that were never committed (toggle-off, clear). */ + discardPending: () => void; +} + +/** + * Axis identity of the chart `ChartDisplay` renders as `chart-0`, for the + * store's axis reset. `graphs` is always `[interactivity, e2e]`, but + * ChartDisplay shows the e2e graph for every non-interactivity x mode + * (`visibleGraphs`), so the rendered chart — not `graphs[0]` — is what the + * rulers were placed on. The x mode itself is part of the identity as well: + * the derived agentic modes (e2e-normalized interactivity, …) override the + * e2e graph's `x_scale_field` inside ChartDisplay only, so here the same + * definition still reads `_e2el` for those modes. Percentile changes + * are already encoded in `x_scale_field`. Null while no graph exists. + */ +export function persistedPerfRulerAxisKey( + graphs: readonly { chartDefinition: { chartType: string; x_scale_field: string } }[], + xAxisMode: string, + yAxisMetric: string, +): string | null { + const wantedType = xAxisMode === 'interactivity' ? 'interactivity' : 'e2e'; + const graph = + graphs.find((candidate) => candidate.chartDefinition.chartType === wantedType) ?? graphs[0]; + if (!graph) return null; + return perfRulerAxisMetricKey(`${xAxisMode}:${graph.chartDefinition.x_scale_field}`, yAxisMetric); +} + +/** Provided by `InferenceProvider`; exported for chart component tests. */ +export const PerfRulerStoreContext = createContext(undefined); + +/** The persisted-ruler store, or undefined outside `InferenceProvider`. */ +export function usePerfRulerStore(): PerfRulerStore | undefined { + return useContext(PerfRulerStoreContext); +} + +/** + * Owns the persisted perf-ruler state. Exported so component tests can host a + * real store around a chart without the full provider. + * + * `axisMetricKey` is the persisted chart's axis identity + * ({@link persistedPerfRulerAxisKey}), or null while no chart definition exists. + * The axis reset runs HERE, not through `usePerfRulerAxisReset` in the chart: + * that hook adjusts state during the chart's render, which is only legal for + * the chart's own state — updating a provider's state from a child's render + * is a React error. Same semantics: a change of either axis metric clears + * committed rulers (redrawn curves would give a ratio nobody placed) and + * discards pending ones (they were placed on the old axes). The null → key + * transition on first data is not a change, so share-link rulers survive + * the load; the x-mode fallback for fixed sequences also settles before any + * chart definition exists. + */ +export function usePerfRulerStoreValue( + chartId: string, + initialSerialized: string | undefined, + axisMetricKey: string | null, +): PerfRulerStore { + const [state, setState] = useState(EMPTY_PERF_RULER_STATE); + const [pending, setPending] = useState(() => { + const parsed = parsePerfRulers(initialSerialized).rulers; + return parsed.length > 0 ? parsed : null; + }); + // `interactivity_perf_ruler_shared_load` fires once per store — once per + // opened link — with the number of rulers the link carried, the first time + // any of them renders. Rulers commit per curve arrival (below), so a + // per-commit event would count one link several times with partial counts. + const linkRulerCountRef = useRef(pending?.length ?? 0); + const sharedLoadReportedRef = useRef(false); + + const [appliedAxisMetricKey, setAppliedAxisMetricKey] = useState(axisMetricKey); + if (axisMetricKey !== null && axisMetricKey !== appliedAxisMetricKey) { + setAppliedAxisMetricKey(axisMetricKey); + if (appliedAxisMetricKey !== null) { + setState(clearPerfRulers); + setPending(null); + } + } + + const commitPending = useCallback( + ( + resolved: readonly PerfRulerMeasurement[], + remaining: readonly PerfRulerMeasurement[] | null, + ) => { + if (resolved.length > 0) { + setState((prev) => { + // Fresh ids from the live counter: a ruler placed by hand before the + // share-link rulers resolved must keep its own join key. + const rulers = [ + ...prev.rulers, + ...resolved.map((ruler, index) => ({ ...ruler, id: prev.nextId + index })), + ]; + while (rulers.length > MAX_PERF_RULERS) rulers.shift(); + return { rulers, draft: prev.draft, nextId: prev.nextId + resolved.length }; + }); + if (!sharedLoadReportedRef.current) { + sharedLoadReportedRef.current = true; + track('interactivity_perf_ruler_shared_load', { count: linkRulerCountRef.current }); + } + } + setPending(remaining); + }, + [], + ); + const discardPending = useCallback(() => setPending(null), []); + + return useMemo( + () => ({ chartId, state, setState, pending, commitPending, discardPending }), + [chartId, state, pending, commitPending, discardPending], + ); +} diff --git a/packages/app/src/components/inference/power-telemetry-dialog.tsx b/packages/app/src/components/inference/power-telemetry-dialog.tsx new file mode 100644 index 000000000..e20718a55 --- /dev/null +++ b/packages/app/src/components/inference/power-telemetry-dialog.tsx @@ -0,0 +1,51 @@ +'use client'; + +import type { InferenceData } from '@/components/inference/types'; +import { PowerTelemetryView } from '@/components/inference/agentic-point/power-telemetry-view'; +import { + Dialog, + DialogContent, + DialogDescription, + DialogHeader, + DialogTitle, +} from '@/components/ui/dialog'; +import { isPersistedBenchmarkId } from '@/lib/benchmark-id'; +import { useLocale } from '@/lib/use-locale'; + +const STRINGS = { + en: { point: 'Benchmark point', concurrency: 'Concurrency' }, + zh: { point: '基准测试数据点', concurrency: '并发数' }, +} as const; + +interface Props { + point: InferenceData; + onOpenChange: (open: boolean) => void; +} + +export function PowerTelemetryDialog({ point, onOpenChange }: Props) { + const t = STRINGS[useLocale()]; + if (!isPersistedBenchmarkId(point.id)) return null; + + return ( + + + + PowerX + + {t.point} #{point.id} · {point.hwKey} · {point.precision.toUpperCase()} ·{' '} + {t.concurrency} {point.conc} + + + + + + ); +} diff --git a/packages/app/src/components/inference/types.ts b/packages/app/src/components/inference/types.ts index c9f9e6c96..cebfb7336 100644 --- a/packages/app/src/components/inference/types.ts +++ b/packages/app/src/components/inference/types.ts @@ -6,6 +6,7 @@ import type { Model, Sequence } from '@/lib/data-mappings'; import type { PowerTier } from '@/lib/power-tier'; import type { SystemPowerEstimate } from '@/lib/modeled-system-power'; import type { MetricKey } from './metric-registry'; +import type { PowerBasis } from '@/lib/power-basis'; export type { WorkerPower }; @@ -347,8 +348,67 @@ export interface InferenceData extends Partial void; setScaleType: (type: 'auto' | 'linear' | 'log') => void; + setPowerCompare: (mode: PowerCompare) => void; setQuickFilterVendors: (vendors: string[]) => void; setQuickFilterFrameworks: (frameworks: string[]) => void; setQuickFilterDeployment: (modes: DeploymentMode[]) => void; diff --git a/packages/app/src/components/inference/ui/ChartControls.tsx b/packages/app/src/components/inference/ui/ChartControls.tsx index 203275706..726a6918c 100644 --- a/packages/app/src/components/inference/ui/ChartControls.tsx +++ b/packages/app/src/components/inference/ui/ChartControls.tsx @@ -61,6 +61,7 @@ import { MetricExplanation } from './MetricExplanation'; import { PowerMetricAvailability } from './PowerMetricAvailability'; import { MeasuredMetricControls } from './MeasuredMetricControls'; import { + changeMeasuredMetricConfig, getMeasuredMetricConfig, MEASURED_METRIC_DEFAULTS, type MeasuredMetricFamily, @@ -230,6 +231,7 @@ export default function ChartControls({ selectedXAxisMetric, selectedXAxisMode, scaleType, + powerCompare, } = useInferenceDisplay(); const { setSelectedModel, @@ -242,6 +244,7 @@ export default function ChartControls({ setSelectedDateRange, setSelectedXAxisMetric, setScaleType, + setPowerCompare, } = useInferenceActions(); // Y-axis options come from the canonical registry and need no API data. @@ -335,10 +338,14 @@ export default function ChartControls({ if (!config) return [option]; if (seen.has(config.family)) return []; seen.add(config.family); + // Keep the selected boundary (and other dimensions) when hopping between + // the power and energy families; fall back to the family default otherwise. const value = selectedConfig?.family === config.family ? selectedYAxisMetric - : MEASURED_METRIC_DEFAULTS[config.family]; + : selectedConfig + ? changeMeasuredMetricConfig(selectedYAxisMetric, { family: config.family }) + : MEASURED_METRIC_DEFAULTS[config.family]; return [ { value, @@ -571,6 +578,8 @@ export default function ChartControls({
`vs. ${word} Time To First Token`, vsE2eLatency: (pctl?: string) => pctl ? `vs. ${pctl} End-to-end Latency` : 'vs. End-to-end Latency', @@ -166,6 +181,15 @@ const STRINGS = { noChartData: '当前模型、场景与筛选条件下没有匹配的基准测试数据。请调整上方筛选条件查看结果。', noSystemPowerData: '当前选择没有可用的系统功耗估算。请选择 8K / 1K 场景;估算仅覆盖 GPU 遥测已验证、硬件受支持、八卡机箱位置已知的运行。存在遥测数据时,仍可单独查看 GPU 实测功耗。', + noUtilityModeledData: + '当前选择没有可用的数据中心建模数值。该边界需要 8K / 1K 场景、已验证的 GPU 遥测,且硬件在机箱功耗模型覆盖范围内(不含 NVL72 系统)。可切换到其他功耗边界以保留数据点。', + powerBasisAssumptions: { + 'gpu-provisioned': + 'GPU 额定边界 · 功率取硬件注册表中每 GPU 的额定 TDP,因此每种硬件的功率曲线为水平线。每输出 token 能耗 = TDP × 分配的 GPU 数 ÷ 整个部署的输出 tok/s;分离式配置将 prefill 与 decode GPU 一并计入。未公布 TDP 的硬件不绘制。', + 'utility-provisioned': + '全电源配置边界 · 功率取硬件注册表中每 GPU 的全电源配置(all-in)市电功率(来源:SemiAnalysis Datacenter Industry Model),因此每种硬件的功率曲线为水平线。每输出 token 能耗 = all-in 功率 × 分配的 GPU 数 ÷ 整个部署的输出 tok/s;分离式配置将 prefill 与 decode GPU 一并计入,这与未加门控的“每输出 token 全电源配置能耗”按 decode GPU 计算不同。', + 'utility-modeled': `数据中心建模边界 · 将 GPU 实测功耗经机箱功耗模型(CPU、DRAM、平台开销、PSU 损耗)推算至市电侧:机箱交流功耗估算 × PUE ${AIR_COOLED_SYSTEM_PUE}(风冷,仅应用一次),再除以实测 GPU 数;每输出 token 能耗按同一比例放大实测能耗。机箱功耗模型版本 ${SYSTEM_POWER_MODEL_REVISION.slice(0, 7)}。仅适用于 8K / 1K、遥测已验证且硬件受支持的运行;NVL72 系统(GB200、GB300)及缺少数值的数据点不绘制。`, + }, vsTtft: (word: string) => `vs. ${word === 'Median' ? '中位' : word} 首 token 延迟(TTFT)`, vsE2eLatency: (pctl?: string) => (pctl ? `vs. ${pctl} 端到端延迟` : 'vs. 端到端延迟'), }, @@ -291,9 +315,19 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean selectedXAxisMode, tokenRevenuePricing, showLineLabels, + powerCompare, } = useInferenceDisplay(); const { setSelectedDates, setSelectedDatesFromRunExpansion, setIsLegendExpanded } = useInferenceActions(); + // The metric key carries the power boundary; the caption discloses it for + // the derived boundaries (there is no separate URL param). + const selectedPowerBasis = getMeasuredMetricConfig(selectedYAxisMetric)?.basis; + // The Measured Power "Timeline" display swaps the scatter body for the + // per-second telemetry traces (PowerTimeline); table view and captions are + // unchanged because the metric key aliases the measured average. + const selectedMeasuredConfig = getMeasuredMetricConfig(selectedYAxisMetric); + const isPowerTimeline = + selectedMeasuredConfig?.family === 'power' && selectedMeasuredConfig.display === 'timeline'; const selectedBenchmarkType: 'single_turn' | 'agentic_traces' = selectedSequence === Sequence.AgenticTraces ? 'agentic_traces' : 'single_turn'; const workflowInfoBenchmarkType = @@ -456,6 +490,7 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean selectedPercentile, tcoBasis, selectedXAxisMode, + powerCompare, }, ); @@ -505,6 +540,7 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean selectedXAxisMetric, selectedE2eXAxisMetric, selectedPercentile, + powerCompare, selectedXAxisMode, tokenRevenuePricing, tcoBasis, @@ -825,7 +861,9 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean

{isModeledSystemPowerConfigKey(selectedYAxisMetric) ? t.noSystemPowerData - : t.noChartData} + : selectedPowerBasis === 'utility-modeled' + ? t.noUtilityModeledData + : t.noChartData}

, ] @@ -1013,6 +1051,8 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean {getSequenceLabel(graph.sequence as Sequence, locale)}{' '} {metricChartTitle(graph.chartDefinition, selectedYAxisMetric, locale)}{' '} {(() => { + // The timeline's x axis is time, not the scatter x metric. + if (isPowerTimeline) return null; const xField = graph.chartDefinition.x_scale_field; if (xField?.endsWith('_ttft')) { const percentile = xField.replace(/_ttft$/u, ''); @@ -1140,6 +1180,15 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean {t.systemPowerAssumptions}

)} + {selectedPowerBasis && selectedPowerBasis !== 'gpu-measured' && ( +

+ {t.powerBasisAssumptions[selectedPowerBasis]} +

+ )} {isUnofficialRun && selectedXAxisMode === 'e2e-normalized-interactivity' && (

@@ -1189,11 +1238,42 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean ); } + if (isPowerTimeline) { + return ( +

+ entry.point), + ]} + overlayData={ + selectUnofficialOverlayForMode( + selectedXAxisMode, + graph.chartDefinition.chartType, + overlayDataByChartType, + ) ?? undefined + } + yLabel={metricLabel( + graph.chartDefinition, + selectedYAxisMetric, + locale, + )} + caption={chartCaption} + /> +
+ ); + } + return isGpuComparison ? ( !point.powerVariant)} xLabel={resolvedXLabel} yLabel={metricLabel(graph.chartDefinition, selectedYAxisMetric, locale)} chartDefinition={graph.chartDefinition} diff --git a/packages/app/src/components/inference/ui/GPUGraph.tsx b/packages/app/src/components/inference/ui/GPUGraph.tsx index 01fc782eb..ac17588d9 100644 --- a/packages/app/src/components/inference/ui/GPUGraph.tsx +++ b/packages/app/src/components/inference/ui/GPUGraph.tsx @@ -1,5 +1,7 @@ 'use client'; +import { useFeatureGate } from '@/lib/use-feature-gate'; +import { getMeasuredMetricConfig } from '@/components/inference/measured-metric-config'; import { track } from '@/lib/analytics'; import { isPersistedBenchmarkId } from '@/lib/benchmark-id'; import { useEphemeralUrlState } from '@/hooks/useUrlState'; @@ -48,6 +50,7 @@ import { chartFrontier, upperPowerEnvelope, isPowerCurveMetric, + isPowerGaugeSeries, isMeasuredPowerCurveMetric, } from '@/components/inference/utils/powerCurves'; import type { @@ -109,6 +112,14 @@ import { } from '@/components/inference/ui/line-label-layer'; import { QuickFiltersDialog } from '@/components/inference/ui/QuickFiltersDialog'; +const PowerTelemetryDialog = dynamic( + () => + import('@/components/inference/power-telemetry-dialog').then( + (module) => module.PowerTelemetryDialog, + ), + { ssr: false }, +); + const FixedSequenceLogDialog = dynamic(() => import('@/components/inference/log-viewer/fixed-sequence-log-dialog').then( (module) => module.FixedSequenceLogDialog, @@ -255,6 +266,11 @@ const GPUGraph = React.memo( setQuickFilterPower, } = useInferenceActions(); const locale = useLocale(); + const featureGateUnlocked = useFeatureGate(); + const showPowerTelemetry = + featureGateUnlocked || getMeasuredMetricConfig(selectedYAxisMetric) !== undefined; + const showPowerTelemetryRef = useRef(showPowerTelemetry); + showPowerTelemetryRef.current = showPowerTelemetry; const legendT = GPU_STRINGS[locale]; const frontierDirection = chartDefinition[ `${selectedYAxisMetric}_roofline` as keyof ChartDefinition @@ -443,10 +459,20 @@ const GPUGraph = React.memo( if (!powerEnvelopeMode) return paretoRooflines; const result: Record = {}; for (const [key, points] of Object.entries(groupedData)) { - result[key] = upperPowerEnvelope(points, chartDefinition.chartType !== 'e2e'); + result[key] = upperPowerEnvelope( + points, + chartDefinition.chartType !== 'e2e', + isPowerGaugeSeries(selectedYAxisMetric, points[0]), + ); } return result; - }, [powerEnvelopeMode, groupedData, paretoRooflines, chartDefinition.chartType]); + }, [ + powerEnvelopeMode, + groupedData, + paretoRooflines, + chartDefinition.chartType, + selectedYAxisMetric, + ]); const boundaryPointKeys = useMemo(() => { const keys = new Set(); @@ -511,6 +537,7 @@ const GPUGraph = React.memo( const logAvailabilityRef = useRef(logAvailability); logAvailabilityRef.current = logAvailability; const [fixedLogPointId, setFixedLogPointId] = useState(null); + const [powerTelemetryPoint, setPowerTelemetryPoint] = useState(null); // Warning annotations for visible series with known upstream issues — // same treatment the scatter view gets, applied to the date-comparison view. @@ -1251,7 +1278,7 @@ const GPUGraph = React.memo( ); } - return ( + const chart = ( ref={chartRef} // Embeds drop the zoom/pan hint line; the host page has its own caption. @@ -1363,6 +1390,7 @@ const GPUGraph = React.memo( yLabel, selectedYAxisMetric, hardwareConfig, + showPowerTelemetry: showPowerTelemetryRef.current, runUrl: d.run_url ? updateRepoUrl(d.run_url) : undefined, hasTrace: isPersistedBenchmarkId(d.id) ? traceAvailabilityRef.current?.[d.id] === true @@ -1424,6 +1452,19 @@ const GPUGraph = React.memo( }); }); } + const powerBtn = tooltipEl.querySelector('[data-action="view-power-telemetry"]'); + if (powerBtn && isPersistedBenchmarkId(d.id)) { + powerBtn.addEventListener('click', (event) => { + event.stopPropagation(); + setPowerTelemetryPoint(d); + chartRef.current?.dismissTooltip(); + track('inference_power_telemetry_opened', { + id: d.id, + hwKey: d.hwKey, + conc: d.conc, + }); + }); + } const logsBtn = tooltipEl.querySelector('[data-action="view-logs"]'); if (logsBtn && typeof d.id === 'number') { logsBtn.addEventListener('click', (event) => { @@ -1694,6 +1735,21 @@ const GPUGraph = React.memo( } /> ); + + return ( + <> + {powerTelemetryPoint === null ? null : ( + { + if (!open) setPowerTelemetryPoint(null); + }} + /> + )} + {chart} + + ); }, ); diff --git a/packages/app/src/components/inference/ui/InferenceTable.test.ts b/packages/app/src/components/inference/ui/InferenceTable.test.ts index bf9ab1b0c..09294dddc 100644 --- a/packages/app/src/components/inference/ui/InferenceTable.test.ts +++ b/packages/app/src/components/inference/ui/InferenceTable.test.ts @@ -1,7 +1,12 @@ import { describe, it, expect } from 'vitest'; +import { createElement } from 'react'; +import { renderToStaticMarkup } from 'react-dom/server'; import type { ChartDefinition, InferenceData } from '@/components/inference/types'; -import { formatInferenceTableNumber } from '@/components/inference/ui/InferenceTable'; +import InferenceTable, { + formatInferenceTableNumber, +} from '@/components/inference/ui/InferenceTable'; +import { expandPowerCompareSeries } from '../utils/power-compare'; import * as inferenceTableModule from './InferenceTable'; import { chartDefinitions } from '../metric-registry'; @@ -42,6 +47,54 @@ function makePoint(overrides: Partial): InferenceData { } describe('InferenceTable sorting logic', () => { + it.each(['roles', 'boundaries'] as const)( + 'renders and sorts each %s comparison by its plotted value', + (mode) => { + const base = makePoint({ + hwKey: 'gb300_dynamo-trt', + y: 708.1, + measuredAvgPower: { y: 708.1, roof: false }, + measuredPrefillAvgPower: { y: 760.442, roof: false }, + measuredDecodeAvgPower: { y: 690.652, roof: false }, + gpuProvisionedWatts: { y: 1400, roof: false }, + utilityProvisionedWatts: { y: 1920, roof: false }, + }); + const points = expandPowerCompareSeries([base], 'y_measuredAvgPower', mode); + const sorted = sortRowsByYMetric(points, chartDefinitions[0], 'y_measuredAvgPower'); + expect(sorted.map((point) => point.y)).toEqual( + mode === 'roles' ? [690.652, 708.1, 760.442] : [708.1, 1400, 1920], + ); + + const html = renderToStaticMarkup( + createElement(InferenceTable, { + data: points, + chartDefinition: chartDefinitions[0], + selectedYAxisMetric: 'y_measuredAvgPower', + }), + ); + const body = html.split('')[1].split('')[0]; + const cells = [...body.matchAll(/]*>(?.*?)<\/tr>/gu)].map((match) => + [...match.groups!.row.matchAll(/]*>(?.*?)<\/td>/gu)].map( + (cell) => cell.groups!.cell, + ), + ); + expect(cells.map((row) => [row[2], row[3]])).toEqual( + mode === 'roles' + ? [ + ['Decode GPUs', '691'], + ['All GPUs', '708'], + ['Prefill GPUs', '760'], + ] + : [ + ['GPU measured', '708'], + ['GPU provisioned (TDP)', '1,400'], + ['Utility provisioned (all-in)', '1,920'], + ], + ); + expect(points.every((point) => point.measuredAvgPower?.y === 708.1)).toBe(true); + }, + ); + it('sorts supported modeled estimates by ascending power', () => { const definition = chartDefinitions[0]; const metric = 'y_modeledChassisPowerPerGpu'; diff --git a/packages/app/src/components/inference/ui/InferenceTable.tsx b/packages/app/src/components/inference/ui/InferenceTable.tsx index b6cdafa7e..0d4e8ab74 100644 --- a/packages/app/src/components/inference/ui/InferenceTable.tsx +++ b/packages/app/src/components/inference/ui/InferenceTable.tsx @@ -7,6 +7,7 @@ import { type DataTableColumn, DataTable } from '@/components/ui/data-table'; import { chipCounts } from '@/lib/chip-counts'; import { getNestedYValue, metricLabel, xAxisLabel } from '@/lib/chart-utils'; import { isModeledSystemPowerConfigKey } from '@/components/inference/metric-registry'; +import { inferPowerCompare, powerSeriesLabel } from '@/components/inference/utils/power-compare'; import { sortRowsByYMetric } from '@/components/inference/ui/inference-table-sort'; import { type Precision, getPrecisionLabel } from '@/lib/data-mappings'; import { getDisplayLabel } from '@/lib/utils'; @@ -43,6 +44,7 @@ export function inferenceTableHeaderLabels( physicalChips: locale === 'zh' ? '物理芯片数' : 'Physical Chips', configuredChips: locale === 'zh' ? '配置中的芯片数' : 'Configured Chip Count', concurrency: locale === 'zh' ? '并发数' : 'Conc', + series: locale === 'zh' ? '系列' : 'Series', yMetric: metricLabel(chartDefinition, selectedYAxisMetric, locale), xMetric: xAxisLabel(chartDefinition, locale), throughput: locale === 'zh' ? '单芯片吞吐量 (tok/s)' : 'Throughput/Chip (tok/s)', @@ -66,6 +68,9 @@ export default function InferenceTable({ () => sortRowsByYMetric(data, chartDefinition, selectedYAxisMetric), [data, chartDefinition, selectedYAxisMetric], ); + // Boundary / role clones (`i_pcompare`) share every config column with their + // base row; the series column is what tells them apart. + const powerCompare = useMemo(() => inferPowerCompare(data), [data]); const columns = useMemo[]>( () => [ @@ -85,6 +90,19 @@ export default function InferenceTable({ className: 'whitespace-nowrap', importance: 'key', }, + ...(powerCompare === 'none' + ? [] + : [ + { + header: headers.series, + cell: (row: InferenceData) => + powerSeriesLabel(row, selectedYAxisMetric, powerCompare, locale), + sortValue: (row: InferenceData) => + powerSeriesLabel(row, selectedYAxisMetric, powerCompare, locale), + className: 'whitespace-nowrap', + importance: 'key' as const, + }, + ]), { header: headers.tensorParallelism, align: 'right', @@ -129,8 +147,12 @@ export default function InferenceTable({ { header: headers.yMetric, align: 'right', - cell: (row) => formatInferenceTableNumber(yPath ? getNestedYValue(row, yPath) : row.y), - sortValue: (row) => (yPath ? getNestedYValue(row, yPath) : row.y), + // Comparison clones keep the source metrics; y holds the plotted role/boundary. + cell: (row) => + formatInferenceTableNumber( + row.powerVariant || !yPath ? row.y : getNestedYValue(row, yPath), + ), + sortValue: (row) => (row.powerVariant || !yPath ? row.y : getNestedYValue(row, yPath)), className: 'tabular-nums', importance: 'key', }, @@ -151,7 +173,7 @@ export default function InferenceTable({ importance: 'key', }, ], - [yPath, headers, showModeledPower], + [yPath, headers, showModeledPower, powerCompare, selectedYAxisMetric, locale], ); return ( diff --git a/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx b/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx index 46f0600eb..0373fc5bd 100644 --- a/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx +++ b/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx @@ -7,15 +7,24 @@ import { SelectTrigger, SelectValue, } from '@/components/ui/select'; +import { track } from '@/lib/analytics'; +import { POWER_BASES, POWER_BASIS_LABELS, type PowerBasis } from '@/lib/power-basis'; import { useLocale } from '@/lib/use-locale'; import { changeMeasuredMetricConfig, getMeasuredMetricConfig, type MeasuredMetricConfigChange, } from '../measured-metric-config'; +import type { PowerCompare } from '../types'; +import { POWER_COMPARE_MODES, powerCompareAvailable } from '../utils/power-compare'; const STRINGS = { en: { + basis: 'Boundary', + basisHelp: + 'Where power is counted. GPU measured: runner telemetry from the GPU boards. GPU provisioned: rated TDP per GPU. Utility provisioned: all-in provisioned utility power per GPU. Utility modeled: measured GPU power carried through the modeled chassis to the utility meter with PUE. Points without a value for the chosen boundary are omitted, never replaced with an estimate.', + basisHint: + 'Derived boundaries report average power per chip across all GPUs, and whole-deployment joules per output token. Changing another setting returns to GPU measured.', scope: 'Scope', scopeHelp: 'All GPUs measures the whole deployment. Prefill and decode select only GPUs serving that role.', @@ -29,7 +38,8 @@ const STRINGS = { roleHint: 'Prefill and decode power support Average only.', display: 'Display', displayHelp: - 'Power per chip in watts, or average power as a percentage of chip TDP. Percent of TDP is available for the all-GPU average only.', + 'Power per chip in watts, average power as a percentage of chip TDP, or the per-second telemetry timeline behind the average. Percent of TDP and Timeline are available for the all-GPU average only.', + timeline: 'Timeline', denominator: 'Per', denominatorHelp: 'Choose the energy denominator. All-GPU energy per input or output token includes the whole deployment; role energy is selected separately under Scope.', @@ -40,8 +50,21 @@ const STRINGS = { unit: 'Unit', unitHelp: 'Energy is shown in joules. Energy per successful query can also be shown in watt-hours.', + compare: 'Compare', + compareHelp: + 'Overlay sibling series on the same points, in the hardware colour with a dash per series. All boundaries: GPU measured, GPU provisioned, utility provisioned and utility modeled. Prefill vs decode: each worker pool next to the whole deployment; on the energy axis the prefill pool is carried onto the output-token axis by the served input:output ratio. Available for the whole-deployment average W/chip and J per output token.', + compareNone: 'Off', + compareBoundaries: 'All boundaries', + compareRoles: 'Prefill vs decode', + compareUnavailable: + 'The comparison is paused for this setting: it needs the whole-deployment average W/chip or J per output token.', }, zh: { + basis: '功耗边界', + basisHelp: + '选择功耗的计量边界。GPU 实测:来自 GPU 板卡的运行器遥测;GPU 额定:每 GPU 的额定 TDP;全电源配置:每 GPU 的全电源配置(all-in)市电功率;数据中心建模:将 GPU 实测功耗经机箱功耗模型推算至市电侧并计入 PUE。所选边界缺少数值的数据点将被省略,不会用估算值替代。', + basisHint: + '推导边界提供全部 GPU 的平均每芯片功率,以及整个部署的每输出 token 能耗;更改其他设置将返回 GPU 实测。', scope: '统计范围', scopeHelp: '全部 GPU 对应整个部署;预填充和解码仅统计承担相应任务的 GPU。', all: '全部 GPU', @@ -54,7 +77,8 @@ const STRINGS = { roleHint: '预填充和解码功率仅支持平均值。', display: '显示方式', displayHelp: - '显示单芯片功率(瓦),或平均功率占芯片 TDP 的百分比。TDP 百分比仅支持全部 GPU 的平均功率。', + '显示单芯片功率(瓦)、平均功率占芯片 TDP 的百分比,或平均值背后的逐秒遥测时间线。TDP 百分比和时间线仅支持全部 GPU 的平均功率。', + timeline: '时间线', denominator: '能耗分母', denominatorHelp: '选择能耗的分母。按输入或输出 token 归一化的全部 GPU 能耗仍包含整个部署;预填充或解码能耗需在统计范围中单独选择。', @@ -64,22 +88,44 @@ const STRINGS = { query: '成功请求', unit: '单位', unitHelp: '能耗以焦耳显示;每个成功请求的能耗也可显示为瓦时。', + compare: '对比', + compareHelp: + '在同一批数据点上叠加同源系列:颜色仍按硬件区分,每个系列用不同虚线表示。全部边界:GPU 实测、GPU 额定、全电源配置、数据中心建模;预填充 vs 解码:各 worker 池与整个部署并列,能耗轴上的预填充能耗按实际服务的输入/输出 token 比折算到每输出 token。仅适用于整个部署的平均 W/芯片和每输出 token 能耗。', + compareNone: '关闭', + compareBoundaries: '全部边界', + compareRoles: '预填充 vs 解码', + compareUnavailable: '当前设置下对比已暂停:需要整个部署的平均 W/芯片或每输出 token 能耗。', }, } as const; export function MeasuredMetricControls({ metric, onChange, + compare = 'none', + onCompareChange, }: { metric: string; onChange: (metric: string) => void; + /** Comparison series overlaid on the metric (`i_pcompare`). */ + compare?: PowerCompare; + onCompareChange?: (mode: PowerCompare) => void; }) { - const t = STRINGS[useLocale()]; + const locale = useLocale(); + const t = STRINGS[locale]; const config = getMeasuredMetricConfig(metric); if (!config) return null; + const compareLabels: Record = { + none: t.compareNone, + boundaries: t.compareBoundaries, + roles: t.compareRoles, + }; + const compareActive = compare !== 'none'; + const compareApplies = powerCompareAvailable(metric, compare); const change = (next: MeasuredMetricConfigChange) => onChange(changeMeasuredMetricConfig(metric, next)); + const basisId = `measured-${config.family}-basis`; const scopeId = `measured-${config.family}-scope`; + const derivedBasis = config.basis !== 'gpu-measured'; const roleScope = config.family === 'energy' ? config.denominator === 'input' @@ -91,9 +137,31 @@ export function MeasuredMetricControls({ return (
+
+ + +
{config.family === 'energy' && (
change({ statistic })} @@ -204,6 +272,13 @@ export function MeasuredMetricControls({ > % TDP + + {t.timeline} +
@@ -237,6 +312,59 @@ export function MeasuredMetricControls({
)} + {onCompareChange && ( +
+ + +
+ )} + {derivedBasis && ( +

+ {t.basisHint} +

+ )} + {compareActive && !compareApplies && ( +

+ {t.compareUnavailable} +

+ )}
); } diff --git a/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx b/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx index a3dd548b0..cad29732b 100644 --- a/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx +++ b/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx @@ -2,8 +2,16 @@ import { useMemo } from 'react'; import { useInferenceData, useInferenceFilters } from '../InferenceContext'; -import { isMeasuredEnergyConfigKey, metricOptionTitle, type MetricKey } from '../metric-registry'; +import { + isMeasuredEnergyConfigKey, + isPowerBasisConfigKey, + metricOptionTitle, + POWER_BASIS_METRIC_CONFIG_KEYS, + type MetricKey, +} from '../metric-registry'; +import { getMeasuredMetricConfig } from '../measured-metric-config'; import type { InferenceData } from '../types'; +import { powerBasisNormalization } from '@/lib/power-basis'; import { matchesQuickFilters } from '../utils/quickFilters'; import { powerMetricAvailability, @@ -33,6 +41,17 @@ const STRINGS = { ambiguous: 'Whole-deployment energy schema unavailable', missing: 'Metric not reported', }, + basisLabels: { + available: 'Value available', + noSpec: 'No published spec for this hardware', + noThroughput: 'No output throughput reported', + noNormalization: 'Whole-deployment GPU count unavailable for this disaggregated row', + noTelemetry: 'No validated GPU telemetry', + invalid: 'Validation failed', + modelWorkload: 'Chassis model covers 8K / 1K only', + modelHardware: 'Hardware not in the chassis power model', + modelUnsupported: 'Chassis model unsupported for this deployment', + }, note: 'A missing verdict does not establish age or validity. Prefill/decode metrics measure separate worker pools. Missing values are never replaced with zero or TDP estimates.', all: 'Availability of all measured metrics', evidence: 'Selected metric: source details', @@ -54,6 +73,17 @@ const STRINGS = { ambiguous: '缺少整个部署的能耗 schema', missing: '未提供此指标', }, + basisLabels: { + available: '有数值', + noSpec: '该硬件没有公开的规格参数', + noThroughput: '未报告输出吞吐量', + noNormalization: '无法确定该分离式部署的 GPU 总数', + noTelemetry: '没有已验证的 GPU 遥测', + invalid: '验证失败', + modelWorkload: '机箱功耗模型仅覆盖 8K / 1K', + modelHardware: '硬件不在机箱功耗模型范围内', + modelUnsupported: '机箱功耗模型不支持此部署', + }, note: '缺少验证结论不能判断数据新旧或有效性。prefill/decode 指标仅衡量独立 worker 池。缺失值不会被替换为零或 TDP 估算值。', all: '所有实测指标的可用性', evidence: '当前指标的来源详情', @@ -61,6 +91,99 @@ const STRINGS = { }, } as const; +/** + * Why a point lacks a derived power boundary (lib/power-basis.ts). Provisioned + * boundaries are spec constants, so their watts exist for any registered + * hardware and their energy additionally needs output throughput plus, for + * disaggregated rows, the whole-deployment GPU count that + * `powerBasisNormalization` recovers only for fixed-sequence runs with integer + * prefill/decode counts. The modeled boundary is measured telemetry carried + * through the chassis model, so it inherits the telemetry verdict and the + * model's own unsupported reasons. + */ +export type PowerBasisAvailabilityState = + | 'available' + | 'noSpec' + | 'noThroughput' + | 'noNormalization' + | 'noTelemetry' + | 'invalid' + | 'modelWorkload' + | 'modelHardware' + | 'modelUnsupported'; + +const POWER_BASIS_AVAILABILITY_STATES: readonly PowerBasisAvailabilityState[] = [ + 'available', + 'noSpec', + 'noThroughput', + 'noNormalization', + 'noTelemetry', + 'invalid', + 'modelWorkload', + 'modelHardware', + 'modelUnsupported', +]; + +const hasFiniteValue = (point: InferenceData, key: MetricKey): boolean => { + const value = point[key]; + return ( + typeof value === 'object' && + value !== null && + 'y' in value && + typeof value.y === 'number' && + Number.isFinite(value.y) + ); +}; + +export function powerBasisState( + point: InferenceData, + configKey: string, +): PowerBasisAvailabilityState { + const key = configKey.replace(/^y_/u, '') as MetricKey; + if (hasFiniteValue(point, key)) return 'available'; + const config = getMeasuredMetricConfig(configKey); + if (config?.basis === 'utility-modeled') { + if (point.power_valid === 0) return 'invalid'; + const model = point.modeledSystemPower; + if (model?.status === 'unsupported') { + if (model.reason === 'workload') return 'modelWorkload'; + if (model.reason === 'hardware') return 'modelHardware'; + if (model.reason === 'telemetry') return 'noTelemetry'; + return 'modelUnsupported'; + } + // A supported model without a plotted value means B1 is absent (B4 follows B1). + return 'noTelemetry'; + } + // Provisioned energy needs the watts sibling plus the whole-deployment + // normalization; the same helper that withheld the value says which half is + // missing, so the explanation cannot drift from the formula. + const wattsKey = ( + config?.basis === 'gpu-provisioned' ? 'gpuProvisionedWatts' : 'utilityProvisionedWatts' + ) satisfies MetricKey; + if (!hasFiniteValue(point, wattsKey)) return 'noSpec'; + const perGpu = point.output_tput_per_gpu; + if (typeof perGpu !== 'number' || !Number.isFinite(perGpu) || perGpu <= 0) return 'noThroughput'; + // Throughput exists, so only the disaggregated GPU count can be missing. + // Chart points may lack the counts an aggregate entry always has; an + // unknown count is exactly the "unavailable" case the helper reports. + const { allocatedGpus } = powerBasisNormalization({ + output_tput_per_gpu: perGpu, + disagg: point.disagg ?? false, + benchmark_type: point.benchmark_type, + num_prefill_gpu: point.num_prefill_gpu ?? Number.NaN, + num_decode_gpu: point.num_decode_gpu ?? Number.NaN, + }); + return allocatedGpus === null ? 'noNormalization' : 'noThroughput'; +} + +function powerBasisAvailability(points: readonly InferenceData[], metric: string) { + const counts = Object.fromEntries( + POWER_BASIS_AVAILABILITY_STATES.map((state) => [state, 0]), + ) as Record; + for (const point of points) counts[powerBasisState(point, metric)]++; + return { metric, counts, available: counts.available, total: points.length }; +} + export function PowerMetricAvailabilityPanel({ points, metric, @@ -74,16 +197,35 @@ export function PowerMetricAvailabilityPanel({ }) { const locale = useLocale(); const t = STRINGS[locale]; - const availability = useMemo(() => powerMetricAvailability(points), [points]); + const availability = useMemo( + () => [ + ...powerMetricAvailability(points), + ...POWER_BASIS_METRIC_CONFIG_KEYS.map((key) => powerBasisAvailability(points, key)), + ], + [points], + ); const selected = availability.find((entry) => entry.metric === metric); if (!selected) return null; + const isBasis = isPowerBasisConfigKey(metric); + const stateOf = (point: InferenceData) => + isBasis ? powerBasisState(point, metric) : powerMetricState(point, metric); + // The two dictionaries overlap on `invalid`; the selected metric, not the + // key, decides which copy applies so the measured strings stay untouched. + const labelOf = (state: PowerAvailabilityState | PowerBasisAvailabilityState) => + isBasis + ? t.basisLabels[state as PowerBasisAvailabilityState] + : t.labels[state as PowerAvailabilityState]; const sources = new Map< string, - { point: InferenceData; state: PowerAvailabilityState; count: number } + { + point: InferenceData; + state: PowerAvailabilityState | PowerBasisAvailabilityState; + count: number; + } >(); for (const point of points) { - const state = powerMetricState(point, metric); - if (state === 'strict') continue; + const state = stateOf(point); + if (state === 'strict' || state === 'available') continue; const key = JSON.stringify([point.hwKey, point.run_url, state, point.power_invalid_reasons]); const group = sources.get(key); if (group) group.count++; @@ -108,7 +250,10 @@ export function PowerMetricAvailabilityPanel({

{Object.entries(selected.counts) .filter(([, count]) => count > 0) - .map(([state, count]) => `${t.labels[state as PowerAvailabilityState]}: ${count}`) + .map( + ([state, count]) => + `${labelOf(state as PowerAvailabilityState | PowerBasisAvailabilityState)}: ${count}`, + ) .join(' · ')}

{[...sources.values()].map(({ point, state, count }, index) => (
  • - {point.hwKey}: {t.labels[state]} ({count}) + {point.hwKey}: {labelOf(state)} ({count}) {point.power_invalid_reasons?.length ? ` · ${point.power_invalid_reasons.join(', ')}` : ''} @@ -221,7 +366,7 @@ export function PowerMetricAvailability({ quickFilters, compareGpuPair, ]); - if (!isMeasuredEnergyConfigKey(metric)) return null; + if (!isMeasuredEnergyConfigKey(metric) && !isPowerBasisConfigKey(metric)) return null; return ( ` end labels would only overlap. */ +const MAX_LABELED_TRACES = 40; +/** Up to this many undrawn configs are named individually; beyond, per hardware. */ +const MAX_LISTED_MISSING = 8; +const CHART_HEIGHT = 600; +const MARGIN = { top: 24, right: 84, bottom: 60, left: 64 }; + +const STRINGS = { + en: { + timeAxis: 'Time axis', + wall: 'Wall clock (UTC)', + elapsed: 'Since start', + xWall: 'Time (UTC)', + xElapsed: 'Time since telemetry start (m:ss)', + perGpu: 'One line per GPU', + perGpuHelp: 'Draw every GPU of a config instead of the mean across its GPUs.', + pools: 'Prefill / decode pools', + poolsHelp: + 'One line per worker-role pool: the summed board power of the prefill GPUs and of the decode GPUs of a config. Dashed references are pool size × rated TDP.', + utilityLines: 'All-in provisioned lines', + utilityHelp: + 'Dashed reference at the all-in provisioned utility power per GPU from the hardware registry (SemiAnalysis Datacenter Industry Model). Off by default because it compresses the traces.', + loading: (runs: number) => + `Loading GPU telemetry for ${runs} run${runs === 1 ? '' : 's'}… (may take a minute)`, + loadError: (runId: string, message: string) => `Run ${runId}: ${message}`, + missing: (missing: number, total: number) => + `${missing} of ${total} measured configs have no telemetry trace and are not drawn.`, + missingReason: { + 'no-source': (count: number) => + `${count} predate per-config telemetry provenance in the benchmark row`, + 'no-run': (count: number) => `${count} carry no workflow run`, + 'run-not-fetched': (count: number) => `${count} come from runs that were not loaded`, + 'not-in-run': (count: number) => + `${count} have no gpu_metrics artifact or power-audit bundle in their run (expired, or another collector)`, + } satisfies Record string>, + missingUndrawn: 'Not drawn', + noTraces: + 'No telemetry traces for the visible hardware. Enable a series in the legend or choose another date.', + noArtifacts: + 'These points predate per-config telemetry artifacts, so no timeline is available for them.', + droppedRuns: (runs: number) => + `Telemetry from ${runs} more run${runs === 1 ? '' : 's'} was not loaded (limit ${POWER_TIMELINE_MAX_RUNS} runs per chart).`, + telemetry: 'Telemetry', + method: + 'One-second means of per-GPU board power (nvidia-smi / amd-smi, or DCGM on Slurm / Dynamo runs) over the whole benchmark job; the emphasized segment is the validated window behind the measured average. Dashed lines: rated TDP per hardware from the hardware registry.', + methodPools: + 'In pool mode each line is the summed power of one worker-role pool (prefill or decode GPUs) and the dashed references are pool size × rated TDP.', + instructions: + 'Shift+Scroll to zoom horizontally · Drag to pan · Double-click to reset · Click a point to pin tooltip', + dismiss: 'Click elsewhere to dismiss', + phase: { + before: 'Before window (startup / warmup)', + window: 'Measurement window', + after: 'After window', + unknown: 'Window not recorded', + } satisfies Record, + meanPerGpu: 'Mean per GPU', + gpus: (count: number) => `${count} GPU${count === 1 ? '' : 's'}`, + min: 'min', + max: 'max', + validated: 'Validated average', + sinceStart: 'since start', + tdp: 'TDP', + allIn: 'all-in', + poolShort: { prefill: 'prefill', decode: 'decode', all: 'all GPUs' } satisfies Record< + PowerPoolRole, + string + >, + yPool: 'GPU pool power (W)', + pool: 'Pool', + poolPower: 'Pool power', + poolTdp: 'pool TDP', + focused: (label: string) => `Focused on ${label}`, + showAll: 'Show all', + unofficialRun: 'Unofficial run', + branch: 'Branch', + viewWorkflow: 'View workflow run', + }, + zh: { + timeAxis: '时间轴', + wall: '实际时刻(UTC)', + elapsed: '相对起点', + xWall: '时间(UTC)', + xElapsed: '距遥测开始的时间(分:秒)', + perGpu: '每个 GPU 一条线', + perGpuHelp: '绘制配置中每个 GPU 的曲线,而不是各 GPU 的平均值。', + pools: '预填充 / 解码 GPU 池', + poolsHelp: + '按 worker 角色分池绘制:每条线是同一配置中预填充 GPU 或解码 GPU 的板卡功耗之和。虚线参考为池内 GPU 数量 × 额定 TDP。', + utilityLines: '全电源配置参考线', + utilityHelp: + '按硬件注册表中每 GPU 的全电源配置(all-in)市电功率绘制虚线参考(SemiAnalysis 数据中心行业模型)。默认关闭,因为它会压缩曲线的纵向分辨率。', + loading: (runs: number) => `正在加载 ${runs} 个运行的 GPU 遥测数据……(可能需要约一分钟)`, + loadError: (runId: string, message: string) => `运行 ${runId}:${message}`, + missing: (missing: number, total: number) => + `${total} 个有实测值的配置中有 ${missing} 个没有遥测曲线,未绘制。`, + missingReason: { + 'no-source': (count: number) => `${count} 个的基准测试行早于按配置记录的遥测来源`, + 'no-run': (count: number) => `${count} 个没有工作流运行信息`, + 'run-not-fetched': (count: number) => `${count} 个来自未加载的运行`, + 'not-in-run': (count: number) => + `${count} 个在其运行中没有 gpu_metrics 产物或 power-audit 数据包(产物已过期,或使用其他采集器)`, + } satisfies Record string>, + missingUndrawn: '未绘制', + noTraces: '当前可见硬件没有遥测曲线。请在图例中启用一个系列或选择其他日期。', + noArtifacts: '这些数据点早于按配置上传的遥测产物,因此没有可用的时间线。', + droppedRuns: (runs: number) => + `另有 ${runs} 个运行的遥测数据未加载(每张图表最多 ${POWER_TIMELINE_MAX_RUNS} 个运行)。`, + telemetry: '遥测来源', + method: + '整个基准测试任务期间每个 GPU 板卡功耗(nvidia-smi / amd-smi,Slurm / Dynamo 运行为 DCGM)的一秒平均值;加粗段为实测平均值所依据的有效测量窗口。虚线:硬件注册表中各硬件的额定 TDP。', + methodPools: + '在 GPU 池模式下,每条线是一个 worker 角色池(预填充或解码 GPU)的功耗总和,虚线参考为池内 GPU 数量 × 额定 TDP。', + instructions: 'Shift+滚轮横向缩放 · 拖动平移 · 双击重置 · 点击数据点固定提示框', + dismiss: '点击其他区域关闭', + phase: { + before: '测量窗口之前(启动 / warmup)', + window: '测量窗口内', + after: '测量窗口之后', + unknown: '未记录测量窗口', + } satisfies Record, + meanPerGpu: '每 GPU 平均', + gpus: (count: number) => `${count} 个 GPU`, + min: '最小', + max: '最大', + validated: '有效平均值', + sinceStart: '距起点', + tdp: 'TDP', + allIn: 'all-in', + poolShort: { prefill: '预填充', decode: '解码', all: '全部 GPU' } satisfies Record< + PowerPoolRole, + string + >, + yPool: 'GPU 池功耗(W)', + pool: 'GPU 池', + poolPower: '池功耗', + poolTdp: '池 TDP', + focused: (label: string) => `聚焦:${label}`, + showAll: '显示全部', + unofficialRun: '非官方运行', + branch: '分支', + viewWorkflow: '查看工作流运行', + }, +} as const; + +type XMode = 'wall' | 'elapsed'; +type LineMode = 'mean' | 'gpu' | 'pool'; + +interface TimelineSample { + trace: PowerTimelineTrace; + color: string; + overlayIndex: number | null; + column: number; + timeMs: number; + /** Data-space x for the active mode: epoch ms (wall) or seconds (elapsed). */ + x: number; + /** Mean watts across the GPUs sampled in the bucket; the pool's summed watts in pool mode. */ + y: number; + min: number; + max: number; + /** GPUs with a sample in the bucket (inside the pool, in pool mode). */ + gpuCount: number; + phase: WindowPhase; + /** The pool this sample sums and its device count, in pool mode. */ + pool?: { role: PowerPoolRole; gpuCount: number }; +} + +interface TracePoint { + x: number; + y: number | null; +} + +interface TracePath { + id: string; + traceKey: string; + hwKey: string; + overlayIndex: number | null; + color: string; + segment: 'full' | 'window'; + width: number; + opacity: number; + points: TracePoint[]; + /** Worker-role pool the line sums, in pool mode. */ + pool?: PowerPoolRole; +} + +interface ReferenceLine { + id: string; + watts: number; + label: string; + color: string; + kind: 'tdp' | 'utility'; + /** Pools the line is sized for, in pool mode (roles sharing one GPU count). */ + pools?: PowerPoolRole[]; +} + +/** Vertical distance between stacked reference labels that share a watts value. */ +const REFERENCE_LABEL_ROW = 13; + +interface TraceLabel { + /** Join key: the trace key, plus the pool role in pool mode. */ + id: string; + traceKey: string; + hwKey: string; + pool?: PowerPoolRole; + color: string; + text: string; + x: number; + y: number; +} + +interface DrawModel { + paths: TracePath[]; + labels: TraceLabel[]; +} + +export interface PowerTimelineProps { + chartId: string; + /** Official points of the chart (display-limit clipped points restored). */ + data: InferenceData[]; + overlayData?: OverlayData; + yLabel: string; + caption?: React.ReactNode; +} + +async function fetchPowerSeries( + request: PowerTimelineRequest, + signal: AbortSignal, +): Promise { + const params = new URLSearchParams({ runId: request.runId, series: 'power' }); + if (request.prefix) params.set('prefix', request.prefix); + const response = await fetch(`/api/gpu-metrics?${params.toString()}`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ sources: request.sources }), + cache: 'no-store', + signal, + }); + const body = (await response.json()) as GpuPowerSeriesResponse | { error: string }; + if (!response.ok) { + throw new Error('error' in body ? body.error : `HTTP ${response.status}`); + } + return body as GpuPowerSeriesResponse; +} + +function formatElapsed(totalSeconds: number): string { + const seconds = Math.max(0, Math.round(totalSeconds)); + const h = Math.floor(seconds / 3600); + const m = Math.floor((seconds % 3600) / 60); + const s = seconds % 60; + const mm = h > 0 ? String(m).padStart(2, '0') : String(m); + return `${h > 0 ? `${h}:` : ''}${mm}:${String(s).padStart(2, '0')}`; +} + +const formatUtcClock = d3.utcFormat('%H:%M:%S'); +const formatUtcDate = d3.utcFormat('%Y-%m-%d'); +/** Pool sums run to thousands of watts; group the digits. */ +const formatWatts = d3.format(',.0f'); + +function baseHardware(hwKey: string): string { + return hwKey.split('_')[0]; +} + +/** + * The pools a trace draws in pool mode: its worker-role pools, or every GPU as + * one pool when the collector assigned no roles (a single-node trace then shows + * its deployment total on the same axis). + */ +function drawnPools(series: GpuPowerSeries): PowerPool[] { + const pools = tracePools(series); + return pools.length > 0 ? pools : [allGpuPool(series)]; +} + +/** SVG dash of a pool line: per role from the comparison palette; `all` stays solid. */ +function poolDash(pool: PowerPoolRole): string | null { + const dash = powerVariantDash({ kind: 'role', id: pool }); + return dash === '' ? null : dash; +} + +interface TraceRow { + id: string; + pool?: PowerPoolRole; + values: (number | null)[]; +} + +/** One polyline's values per line mode: the GPU mean, each GPU, or each pool's sum. */ +function traceRows(series: GpuPowerSeries, lineMode: LineMode): TraceRow[] { + if (lineMode === 'gpu') { + return series.gpus.map((gpu, row) => ({ id: `gpu${gpu}`, values: series.power[row] })); + } + if (lineMode === 'pool') { + return drawnPools(series).map((pool) => ({ + id: `pool:${pool.role}`, + pool: pool.role, + values: series.t.map((_, column) => sumPowerAt(series, pool.rows, column)), + })); + } + return [{ id: 'mean', values: series.t.map((_, column) => meanPowerAt(series, column)) }]; +} + +/** Builds the mean, per-GPU or per-pool polylines plus the window emphasis for one trace. */ +function tracePaths( + trace: PowerTimelineTrace, + color: string, + overlayIndex: number | null, + xMode: XMode, + lineMode: LineMode, +): TracePath[] { + const { series } = trace; + const xOf = (column: number) => + xMode === 'wall' ? bucketTimeMs(series, column) : series.t[column] - series.t[0]; + const rows = traceRows(series, lineMode); + const faint = lineMode === 'gpu' ? 0.22 : 0.32; + const strong = lineMode === 'gpu' ? 0.85 : 1; + const widths = lineMode === 'gpu' ? [1, 1.5] : [1.25, 2.25]; + const paths: TracePath[] = []; + for (const row of rows) { + const full: TracePoint[] = []; + const window: TracePoint[] = []; + row.values.forEach((value, column) => { + const point = { x: xOf(column), y: value }; + full.push(point); + if (windowPhase(trace, bucketTimeMs(series, column)) === 'window') window.push(point); + }); + paths.push({ + id: `${trace.key}:${row.id}:full`, + traceKey: trace.key, + hwKey: trace.point.hwKey, + overlayIndex, + color, + segment: 'full', + width: widths[0], + opacity: faint, + points: full, + pool: row.pool, + }); + if (window.length > 1) { + paths.push({ + id: `${trace.key}:${row.id}:window`, + traceKey: trace.key, + hwKey: trace.point.hwKey, + overlayIndex, + color, + segment: 'window', + width: widths[1], + opacity: strong, + points: window, + pool: row.pool, + }); + } + } + return paths; +} + +/** + * Hover targets of one trace: one stream over all its GPUs (mean watts), or in + * pool mode one stream per pool (summed watts) so the tooltip can name the pool. + */ +function traceSamples( + trace: PowerTimelineTrace, + color: string, + overlayIndex: number | null, + xMode: XMode, + lineMode: LineMode, +): TimelineSample[] { + const { series } = trace; + if (lineMode !== 'pool') { + const rows = series.power.map((_, row) => row); + return sampleRows(trace, color, overlayIndex, xMode, rows, undefined); + } + return drawnPools(series).flatMap((pool) => + sampleRows(trace, color, overlayIndex, xMode, pool.rows, { + role: pool.role, + gpuCount: pool.rows.length, + }), + ); +} + +function sampleRows( + trace: PowerTimelineTrace, + color: string, + overlayIndex: number | null, + xMode: XMode, + rows: readonly number[], + pool: TimelineSample['pool'], +): TimelineSample[] { + const { series } = trace; + const samples: TimelineSample[] = []; + for (let column = 0; column < series.t.length; column++) { + let sum = 0; + let count = 0; + let min = Number.POSITIVE_INFINITY; + let max = Number.NEGATIVE_INFINITY; + for (const row of rows) { + const value = series.power[row]?.[column]; + if (value === null || value === undefined) continue; + sum += value; + count += 1; + if (value < min) min = value; + if (value > max) max = value; + } + // A pool bucket missing a device is a gap, as in `sumPowerAt`, not a dip. + if (count === 0 || (pool && count < rows.length)) continue; + const timeMs = bucketTimeMs(series, column); + samples.push({ + trace, + color, + overlayIndex, + column, + timeMs, + x: xMode === 'wall' ? timeMs : series.t[column] - series.t[0], + y: pool ? sum : sum / count, + min, + max, + gpuCount: count, + phase: windowPhase(trace, timeMs), + pool, + }); + } + return lttbDownsample( + samples, + HIT_POINTS_PER_TRACE, + (sample) => sample.x, + (sample) => sample.y, + ); +} + +type AnyContinuousScale = d3.ScaleLinear; + +function drawTraces( + group: d3.Selection, + xScale: AnyContinuousScale, + yScale: AnyContinuousScale, + model: DrawModel, + highlight: string | null, +): void { + const line = d3 + .line() + .defined((point) => point.y !== null) + .x((point) => xScale(point.x)) + .y((point) => yScale(point.y ?? 0)) + .curve(d3.curveLinear); + const selection = group + .selectAll('path.power-trace') + .data(model.paths, (path) => path.id); + selection.exit().remove(); + selection + .enter() + .append('path') + .attr('class', 'power-trace') + .attr('fill', 'none') + .attr('stroke-linejoin', 'round') + .attr('stroke-linecap', 'round') + .merge(selection) + .attr('data-trace-key', (path) => path.traceKey) + .attr('data-hw', (path) => path.hwKey) + .attr('data-segment', (path) => path.segment) + .attr('data-run-index', (path) => (path.overlayIndex === null ? null : path.overlayIndex)) + .attr('data-pool', (path) => path.pool ?? null) + .attr('stroke-dasharray', (path) => (path.pool ? poolDash(path.pool) : null)) + .attr('stroke', (path) => path.color) + .attr('stroke-width', (path) => path.width) + .attr('opacity', (path) => traceOpacity(path, highlight)) + .attr('d', (path) => line(path.points)); +} + +function traceOpacity(path: TracePath, highlight: string | null): number { + if (highlight === null) return path.opacity; + return highlight === path.hwKey || highlight === path.traceKey + ? Math.min(1, path.opacity + 0.15) + : path.opacity * 0.15; +} + +function drawLabels( + group: d3.Selection, + xScale: AnyContinuousScale, + yScale: AnyContinuousScale, + model: DrawModel, + plotWidth: number, + highlight: string | null, +): void { + const selection = group + .selectAll('text.power-trace-label') + .data(model.labels, (label) => label.id); + selection.exit().remove(); + selection + .enter() + .append('text') + .attr('class', 'power-trace-label') + .attr('font-family', CHART_FONT_SANS) + .attr('font-size', px(CHART_TYPE.dataLabel)) + .attr('font-weight', '600') + .attr('dominant-baseline', 'middle') + .attr('pointer-events', 'none') + .merge(selection) + .attr('data-hw', (label) => label.hwKey) + .attr('data-pool', (label) => label.pool ?? null) + .attr('fill', (label) => label.color) + .attr('opacity', (label) => + highlight === null || highlight === label.hwKey || highlight === label.traceKey ? 1 : 0.2, + ) + .text((label) => label.text) + .each(function (label) { + const x = xScale(label.x); + const y = yScale(label.y); + // Sit just past the last sample; flip inside the plot near the right edge. + const overflow = x + 6 + label.text.length * 6.5 > plotWidth; + d3.select(this) + .attr('text-anchor', overflow ? 'end' : 'start') + .attr('x', overflow ? x - 6 : x + 6) + .attr('y', y); + }); +} + +function drawReferenceLines( + group: d3.Selection, + yScale: AnyContinuousScale, + width: number, + lines: ReferenceLine[], +): void { + group.selectAll('.power-reference').remove(); + const slots = referenceLabelSlots(lines); + lines.forEach((line, index) => { + const y = yScale(line.watts); + const g = group + .append('g') + .attr('class', 'power-reference') + .attr('data-reference', line.kind) + .attr('data-pool', line.pools?.join(' ') ?? null) + .attr('data-watts', line.watts); + g.append('line') + .attr('x1', 0) + .attr('x2', width) + .attr('y1', y) + .attr('y2', y) + .attr('stroke', line.color) + .attr('stroke-width', 1.25) + .attr('stroke-dasharray', line.kind === 'tdp' ? '6,4' : '2,4') + .attr('opacity', 0.9); + g.append('text') + .attr('x', width - 4) + .attr('y', y - 5 - slots[index] * REFERENCE_LABEL_ROW) + .attr('text-anchor', 'end') + .attr('fill', line.color) + .attr('font-family', CHART_FONT_SANS) + .attr('font-size', px(CHART_TYPE.annotation)) + .attr('font-weight', '600') + .text(line.label); + }); +} + +/** Reasons, then the undrawn configs (named when few, counted per hardware when many). */ +function describeMissing( + missing: readonly MissingTrace[], + t: (typeof STRINGS)[keyof typeof STRINGS], + hardwareLabel: (point: InferenceData) => string, +): string { + const reasons = new Map(); + for (const { reason } of missing) reasons.set(reason, (reasons.get(reason) ?? 0) + 1); + const reasonText = [...reasons.entries()] + .map(([reason, count]) => t.missingReason[reason](count)) + .join('; '); + let list: string; + if (missing.length <= MAX_LISTED_MISSING) { + list = missing + .map(({ point }) => `${hardwareLabel(point)} ${traceConfigLabel(point)}`) + .join(' · '); + } else { + const perHardware = new Map(); + for (const { point } of missing) { + const label = hardwareLabel(point); + perHardware.set(label, (perHardware.get(label) ?? 0) + 1); + } + list = [...perHardware.entries()].map(([label, count]) => `${label} ×${count}`).join(' · '); + } + return `${reasonText}. ${t.missingUndrawn}: ${list}`; +} + +export default function PowerTimeline({ + chartId, + data, + overlayData, + yLabel, + caption, +}: PowerTimelineProps) { + const locale = useLocale(); + const t = STRINGS[locale]; + const { hardwareConfig, hwTypesWithData } = useInferenceData(); + const { activeHwTypes, selectedPrecisions, quickFilters } = useInferenceFilters(); + const { isLegendExpanded, highContrast } = useInferenceDisplay(); + const { setBestPerSku, toggleHwType, setIsLegendExpanded } = useInferenceActions(); + const { + unofficialRunInfos, + runIndexByUrl, + activeOverlayHwTypes, + localOfficialOverride, + setUnifiedOverlaySelection, + } = useUnofficialRun(); + + const [xModeChoice, setXModeChoice] = useState(null); + const [lineMode, setLineMode] = useState('mean'); + const [showUtility, setShowUtility] = useState(false); + const [highlight, setHighlight] = useState(null); + /** Trace a "View power trace" deep link asked for, once the join has produced it. */ + const [focusKey, setFocusKey] = useState(null); + /** The deep-link request, read once on mount; `undefined` until read, `null` once honoured. */ + const requestedFocusRef = useRef(undefined); + + // The chart's point list still carries every precision, quick-filtered rows + // and rows without a validated average (ScatterGraph applies those gates at + // draw time); only rows that plot on the measured-average axis for the + // current selection have a trace to look up. + const plotsHere = useCallback( + (point: InferenceData) => + point.measuredPowerTimeline !== undefined && + selectedPrecisions.includes(point.precision) && + matchesQuickFilters(point, quickFilters), + [selectedPrecisions, quickFilters], + ); + const measuredData = useMemo(() => data.filter(plotsHere), [data, plotsHere]); + const overlayPoints = useMemo( + () => (overlayData?.data ?? []).filter(plotsHere), + [overlayData, plotsHere], + ); + const overlayPointSet = useMemo(() => new Set(overlayPoints), [overlayPoints]); + const allPoints = useMemo( + () => [...measuredData, ...overlayPoints], + [measuredData, overlayPoints], + ); + + const hwKeysInData = useMemo( + () => + [...new Set(measuredData.map((point) => point.hwKey))].toSorted( + (a, b) => getModelSortIndex(a) - getModelSortIndex(b) || a.localeCompare(b), + ), + [measuredData], + ); + const stableHcKeys = useMemo(() => [...hwTypesWithData], [hwTypesWithData]); + const activeOfficialKeys = useMemo(() => [...activeHwTypes], [activeHwTypes]); + const { resolveColor, getCssColor } = useThemeColors({ + highContrast, + identifiers: hwKeysInData, + activeKeys: activeOfficialKeys, + hcKeys: stableHcKeys, + }); + + // ── Telemetry fetch: one request per workflow run ────────────────────────── + const requests = useMemo(() => planPowerTimelineRequests(allPoints), [allPoints]); + // The deep-link request is read once, before planning, so its run is fetched + // even when the chart spans more runs than the cap. + if (requestedFocusRef.current === undefined) { + requestedFocusRef.current = consumePowerTraceFocus(); + } + const focusRunRef = useRef(traceKeyRunId(requestedFocusRef.current)); + // Overlay runs were requested explicitly (`?unofficialrun=`), so they take + // the cap's slots before official runs; the deep-linked run still goes first. + const overlayRunIds = useMemo( + () => + new Set( + overlayPoints + .map((point) => runIdFromUrl(point.run_url)) + .filter((runId): runId is string => runId !== null), + ), + [overlayPoints], + ); + const fetchedRequests = useMemo( + () => + prioritizeRun(prioritizeRuns(requests, overlayRunIds), focusRunRef.current).slice( + 0, + POWER_TIMELINE_MAX_RUNS, + ), + [requests, overlayRunIds], + ); + const droppedRuns = requests.length - fetchedRequests.length; + // React Query structurally shares the combined result, so `resolved` keeps + // its identity until a run's data actually changes. That gives the response + // map a fixed-shape memo input; a per-query spread would change the deps + // array length whenever runs enter or leave the plot, which React rejects. + const combineQueries = useCallback( + (results: UseQueryResult[]) => ({ + loadingRuns: results.filter((query) => query.isPending).length, + errors: fetchedRequests.flatMap((request, index) => { + const error = results[index]?.error; + return error instanceof Error ? [{ request, error }] : []; + }), + resolved: fetchedRequests.flatMap((request, index) => { + const response = results[index]?.data; + return response ? [[request.runId, response] as const] : []; + }), + }), + [fetchedRequests], + ); + const { loadingRuns, errors, resolved } = useQueries({ + queries: fetchedRequests.map((request) => ({ + queryKey: ['power-timeline', request.runId, request.prefix, request.sources] as const, + queryFn: ({ signal }: { signal: AbortSignal }) => fetchPowerSeries(request, signal), + staleTime: 0, + refetchOnWindowFocus: true, + retry: 1, + })), + combine: combineQueries, + }); + const responses = useMemo(() => new Map(resolved), [resolved]); + + const { traces, missing } = useMemo( + () => joinPowerTimeline(allPoints, responses), + [allPoints, responses], + ); + const hasAnyArtifact = requests.length > 0; + + // Honour the deep link once its trace exists; a disaggregated trace opens in + // pool mode because its prefill / decode split is what the reader came for. + useEffect(() => { + const requested = requestedFocusRef.current; + if (!requested) return; + const trace = traces.find((entry) => entry.key === requested); + if (!trace) return; + requestedFocusRef.current = null; + setFocusKey(trace.key); + if (tracePools(trace.series).length > 0) setLineMode('pool'); + }, [traces]); + + useEffect(() => { + if (loadingRuns > 0 || responses.size === 0) return; + track('inference_power_timeline_loaded', { + traces: traces.length, + missing: missing.length, + runs: responses.size, + }); + }, [loadingRuns, responses.size, traces.length, missing.length]); + + // ── Visible traces and their colours ──────────────────────────────────────── + const colorForTrace = useCallback( + (trace: PowerTimelineTrace): { color: string; overlayIndex: number | null } => { + if (overlayPointSet.has(trace.point)) { + const index = overlayRunIndex(trace.point.run_url ?? null, runIndexByUrl); + return { color: overlayRunColor(index), overlayIndex: index }; + } + return { color: getCssColor(resolveColor(trace.point.hwKey)), overlayIndex: null }; + }, + [overlayPointSet, runIndexByUrl, getCssColor, resolveColor], + ); + // Same visibility source as ScatterGraph: an overlay session may hold a + // local official selection that has not been written back to the filters. + const officialHwTypes = localOfficialOverride ?? activeHwTypes; + // With an overlay loaded the chart reads localOfficialOverride, so a legend + // click must write the unified selection the way ScatterGraph does; the + // context's toggleHwType would change activeHwTypes with no visible effect. + const handleToggleHwType = useCallback( + (key: string) => { + if (!overlayData) { + toggleHwType(key); + return; + } + setBestPerSku(false, { applySelection: false }); + const official = new Set([...officialHwTypes].filter((hw) => hwTypesWithData.has(hw))); + setUnifiedOverlaySelection( + computeToggle(official, key, hwTypesWithData), + activeOverlayHwTypes, + ); + }, + [ + overlayData, + toggleHwType, + setBestPerSku, + officialHwTypes, + hwTypesWithData, + setUnifiedOverlaySelection, + activeOverlayHwTypes, + ], + ); + const visibleTraces = useMemo( + () => + traces.filter((trace) => + overlayPointSet.has(trace.point) + ? activeOverlayHwTypes.has(trace.point.hwKey) + : officialHwTypes.has(trace.point.hwKey), + ), + [traces, overlayPointSet, activeOverlayHwTypes, officialHwTypes], + ); + // Focus follows visibility: hiding the focused hardware in the legend lifts + // the dimming and the chip instead of dimming everything with nothing lit. + const focusedTrace = useMemo( + () => visibleTraces.find((trace) => trace.key === focusKey) ?? null, + [visibleTraces, focusKey], + ); + /** Legend hover wins over the deep-link focus while it lasts. */ + const activeHighlight = highlight ?? focusedTrace?.key ?? null; + const visibleRunCount = useMemo( + () => new Set(visibleTraces.map((trace) => trace.runId)).size, + [visibleTraces], + ); + const xMode: XMode = xModeChoice ?? (visibleRunCount <= 1 ? 'wall' : 'elapsed'); + + // The pools switch is offered only where a visible trace carries worker roles; + // pool mode left without one would draw deployment totals with no way back. + const hasPools = useMemo( + () => visibleTraces.some((trace) => tracePools(trace.series).length > 0), + [visibleTraces], + ); + useEffect(() => { + if (lineMode === 'pool' && !hasPools && visibleTraces.length > 0) setLineMode('mean'); + }, [lineMode, hasPools, visibleTraces.length]); + + const model = useMemo(() => { + const paths: TracePath[] = []; + const labels: TraceLabel[] = []; + for (const trace of visibleTraces) { + const { color, overlayIndex } = colorForTrace(trace); + const tracePathSet = tracePaths(trace, color, overlayIndex, xMode, lineMode); + paths.push(...tracePathSet); + if (visibleTraces.length > MAX_LABELED_TRACES) continue; + // One end label per trace; per pool in pool mode, so the role reads off the line. + const groups: { pool?: PowerPoolRole; paths: TracePath[] }[] = + lineMode === 'pool' + ? drawnPools(trace.series).map((pool) => ({ + pool: pool.role, + paths: tracePathSet.filter((path) => path.pool === pool.role), + })) + : [{ paths: tracePathSet }]; + for (const group of groups) { + const anchor = + group.paths.find((path) => path.segment === 'window') ?? + group.paths.find((path) => path.segment === 'full'); + const last = anchor?.points.filter((point) => point.y !== null).at(-1); + if (!last || last.y === null) continue; + labels.push({ + id: group.pool ? `${trace.key}:${group.pool}` : trace.key, + traceKey: trace.key, + hwKey: trace.point.hwKey, + pool: group.pool, + color, + text: group.pool + ? `c${trace.point.conc} · ${t.poolShort[group.pool]}` + : `c${trace.point.conc}`, + x: last.x, + y: last.y, + }); + } + } + return { paths, labels }; + }, [visibleTraces, colorForTrace, xMode, lineMode, t]); + + const samples = useMemo( + () => + visibleTraces.flatMap((trace) => { + const { color, overlayIndex } = colorForTrace(trace); + return traceSamples(trace, color, overlayIndex, xMode, lineMode); + }), + [visibleTraces, colorForTrace, xMode, lineMode], + ); + + // Rated references: per hardware in mean / per-GPU modes; per (hardware, + // pool size) in pool mode, scaled to the pool so the summed line and its + // ceiling share the axis. Roles of one hardware that hold the same number + // of GPUs share a ceiling and draw as one line (`prefill / decode ×16`). + const referenceLines = useMemo(() => { + const lines: ReferenceLine[] = []; + const pushLines = (base: string, color: string, pool?: PoolSizeGroup) => { + const specs = HW_REGISTRY[base]; + if (!specs) return; + const label = specs.label ?? base.toUpperCase(); + const size = pool?.size ?? 1; + const id = pool ? `${base}:${pool.roles.join('+')}:${pool.size}` : base; + const name = pool + ? `${label} ${pool.roles.map((role) => t.poolShort[role]).join(' / ')} ×${pool.size}` + : label; + if (specs.tdp > 0) { + lines.push({ + id: `tdp:${id}`, + watts: specs.tdp * size, + label: `${name} ${t.tdp} ${specs.tdp * size} W`, + color, + kind: 'tdp', + pools: pool?.roles, + }); + } + if (showUtility && specs.power > 0) { + const watts = pool ? Math.round(specs.power * 1000) * size : specs.power * 1000; + lines.push({ + id: `utility:${id}`, + watts, + label: `${name} ${t.allIn} ${Math.round(watts)} W`, + color, + kind: 'utility', + pools: pool?.roles, + }); + } + }; + // First trace of a hardware sets the reference colour, as before. + const perBase = new Map(); + for (const trace of visibleTraces) { + const base = baseHardware(trace.point.hwKey); + if (!perBase.has(base)) perBase.set(base, { color: colorForTrace(trace).color, pools: [] }); + if (lineMode === 'pool') perBase.get(base)!.pools.push(...drawnPools(trace.series)); + } + for (const [base, { color, pools }] of perBase) { + if (lineMode !== 'pool') { + pushLines(base, color); + continue; + } + for (const group of groupPoolsBySize(pools)) pushLines(base, color, group); + } + return lines; + }, [visibleTraces, colorForTrace, showUtility, lineMode, t]); + + // ── Scales ───────────────────────────────────────────────────────────────── + const xDomain = useMemo<[number, number]>(() => { + let min = Number.POSITIVE_INFINITY; + let max = Number.NEGATIVE_INFINITY; + for (const path of model.paths) { + if (path.segment !== 'full') continue; + for (const point of path.points) { + if (point.x < min) min = point.x; + if (point.x > max) max = point.x; + } + } + if (!Number.isFinite(min) || !Number.isFinite(max)) { + return xMode === 'wall' ? [Date.UTC(2026, 0, 1), Date.UTC(2026, 0, 1, 0, 10)] : [0, 600]; + } + return min === max ? [min, max + (xMode === 'wall' ? 60_000 : 60)] : [min, max]; + }, [model.paths, xMode]); + const yDomain = useMemo<[number, number]>(() => { + let max = 0; + for (const path of model.paths) { + for (const point of path.points) if (point.y !== null && point.y > max) max = point.y; + } + for (const line of referenceLines) if (line.watts > max) max = line.watts; + return [0, max > 0 ? max * 1.06 : 100]; + }, [model.paths, referenceLines]); + + const xTickFormat = useMemo(() => { + if (xMode === 'elapsed') return (value: d3.AxisDomain) => formatElapsed(Number(value)); + const span = xDomain[1] - xDomain[0]; + const crossesDate = formatUtcDate(new Date(xDomain[0])) !== formatUtcDate(new Date(xDomain[1])); + const format = d3.utcFormat( + crossesDate ? '%m/%d %H:%M' : span < 3 * 60_000 ? '%H:%M:%S' : '%H:%M', + ); + return (value: d3.AxisDomain) => + format(value instanceof Date ? value : new Date(Number(value))); + }, [xMode, xDomain]); + + // ── Layers ───────────────────────────────────────────────────────────────── + const highlightRef = useRef(activeHighlight); + highlightRef.current = activeHighlight; + const layers = useMemo[]>( + () => [ + { + type: 'custom', + key: 'power-reference-lines', + render: (group, ctx) => { + drawReferenceLines(group, ctx.yScale as AnyContinuousScale, ctx.width, referenceLines); + }, + }, + { + type: 'custom', + key: 'power-traces', + render: (group, ctx: RenderContext) => { + drawTraces( + group, + ctx.xScale as AnyContinuousScale, + ctx.yScale as AnyContinuousScale, + model, + highlightRef.current, + ); + }, + onZoom: (group, ctx: ZoomContext) => { + drawTraces( + group, + ctx.newXScale as AnyContinuousScale, + ctx.newYScale as AnyContinuousScale, + model, + highlightRef.current, + ); + }, + }, + { + type: 'point', + key: 'power-hit-points', + data: samples, + config: { + getCx: () => 0, + getCy: () => 0, + getX: (sample) => sample.x, + getY: (sample) => sample.y, + getColor: (sample) => sample.color, + getRadius: () => 2, + // Pool streams of one trace share columns; the role keeps their keys apart. + keyFn: (sample) => + sample.pool + ? `${sample.trace.key}:${sample.pool.role}:${sample.column}` + : `${sample.trace.key}:${sample.column}`, + maxPoints: Number.POSITIVE_INFINITY, + }, + }, + { + type: 'custom', + key: 'power-trace-labels', + render: (group, ctx: RenderContext) => { + drawLabels( + group, + ctx.xScale as AnyContinuousScale, + ctx.yScale as AnyContinuousScale, + model, + ctx.width, + highlightRef.current, + ); + }, + onZoom: (group, ctx: ZoomContext) => { + drawLabels( + group, + ctx.newXScale as AnyContinuousScale, + ctx.newYScale as AnyContinuousScale, + model, + ctx.width, + highlightRef.current, + ); + }, + }, + ], + [referenceLines, model, samples], + ); + + const onDisplayUpdate = useCallback( + (ctx: RenderContext) => { + const root = d3.select(ctx.layout.svg.node() as SVGSVGElement); + root + .selectAll('path.power-trace') + .attr('opacity', (path) => traceOpacity(path, activeHighlight)); + root + .selectAll('text.power-trace-label') + .attr('opacity', (label) => + activeHighlight === null || + activeHighlight === label.hwKey || + activeHighlight === label.traceKey + ? 1 + : 0.2, + ); + }, + [activeHighlight], + ); + + const hardwareLabel = useCallback( + (point: InferenceData): string => { + const config = overlayPointSet.has(point) + ? overlayData?.hardwareConfig[point.hwKey] + : hardwareConfig[point.hwKey]; + return config ? getDisplayLabel(config) : point.hwKey; + }, + [overlayPointSet, overlayData, hardwareConfig], + ); + + const tooltipContent = useCallback( + (sample: TimelineSample, isPinned: boolean) => { + const { trace } = sample; + const point = trace.point; + const overlayInfo = + sample.overlayIndex === null ? null : unofficialRunInfos[sample.overlayIndex]; + const tdp = HW_REGISTRY[baseHardware(point.hwKey)]?.tdp ?? 0; + const elapsed = formatElapsed((sample.timeMs - trace.series.startMs) / 1000); + const clock = `${formatUtcClock(new Date(sample.timeMs))} UTC`; + const time = + xMode === 'wall' ? `${clock} · +${elapsed} ${t.sinceStart}` : `+${elapsed} · ${clock}`; + const colon = locale === 'zh' ? ':' : ':'; + const validated = point.measuredAvgPower?.y; + const { pool } = sample; + const readings = pool + ? `
    ${t.pool}${colon} ${t.poolShort[pool.role]} · ${t.gpus(pool.gpuCount)}
    +
    ${t.poolPower}${colon} ${formatWatts(sample.y)} W${ + tdp > 0 + ? ` (${((sample.y / (tdp * pool.gpuCount)) * 100).toFixed(0)}% ${t.poolTdp})` + : '' + }
    +
    ${t.meanPerGpu}${colon} ${(sample.y / sample.gpuCount).toFixed(1)} W · ${t.min} ${sample.min.toFixed(1)} W · ${t.max} ${sample.max.toFixed(1)} W
    ` + : `
    ${t.meanPerGpu}${colon} ${sample.y.toFixed(1)} W${ + tdp > 0 + ? ` (${((sample.y / tdp) * 100).toFixed(0)}% ${t.tdp})` + : '' + }
    +
    ${t.gpus(sample.gpuCount)} · ${t.min} ${sample.min.toFixed(1)} W · ${t.max} ${sample.max.toFixed(1)} W
    `; + return `
    + ${isPinned ? `
    ${t.dismiss}
    ` : ''} +
    ${hardwareLabel(point)} · ${traceConfigLabel(point)}${ + overlayInfo ? ` · ✕ ${overlayInfo.branch || `run ${overlayInfo.id}`}` : '' + }
    +
    ${time}
    + ${readings} +
    ${t.phase[sample.phase]}
    + ${ + typeof validated === 'number' + ? `
    ${t.validated}${colon} ${validated.toFixed(1)} W
    ` + : '' + } +
    `; + }, + [unofficialRunInfos, xMode, t, locale, hardwareLabel], + ); + + // ── Legend ───────────────────────────────────────────────────────────────── + const legendItems = useMemo(() => { + const overlayItems = + overlayData && unofficialRunInfos.length > 0 + ? unofficialRunInfos + .map((info, index) => { + const hasPoints = overlayPoints.some( + (point) => overlayRunIndex(point.run_url ?? null, runIndexByUrl) === index, + ); + if (!hasPoints) return null; + const branch = info.branch || `run ${info.id}`; + return { + name: `✕ unofficial-run-${info.id}`, + label: `✕ ${branch}`, + color: overlayRunColor(index), + title: `${t.unofficialRun}: ${branch}`, + isHighlighted: true, + hw: `overlay-run-${info.id}`, + isActive: true, + isRemovable: false, + onClick: () => {}, + tooltip: ( +
    +
    {t.unofficialRun}
    +
    + {t.branch}: {branch} +
    + {info.url && ( + + {t.viewWorkflow} + + )} +
    + ), + }; + }) + .filter((item): item is NonNullable => item !== null) + : []; + const officialItems = hwKeysInData + .filter((key) => hwTypesWithData.has(key) && hardwareConfig[key]) + .map((key) => { + const config = hardwareConfig[key]; + return { + name: config.name, + label: getDisplayLabel(config), + color: resolveColor(key), + title: config.gpu, + hw: key, + isActive: officialHwTypes.has(key), + onClick: () => { + handleToggleHwType(key); + track('latency_hw_type_toggled', { hw: key }); + }, + tooltip: null, + }; + }); + return [...overlayItems, ...officialItems]; + }, [ + overlayData, + unofficialRunInfos, + overlayPoints, + runIndexByUrl, + hwKeysInData, + hwTypesWithData, + hardwareConfig, + resolveColor, + officialHwTypes, + handleToggleHwType, + t, + ]); + + // Per-GPU and pools are two views of the same lines, so either switch turns + // the other off; both fall back to the mean. + const chooseLineMode = (next: LineMode) => { + setLineMode(next); + track('inference_power_timeline_lines_changed', { lines: next }); + }; + const switches: LegendSwitchConfig[] = [ + { + id: 'power-timeline-per-gpu', + label: t.perGpu, + checked: lineMode === 'gpu', + onCheckedChange: (checked) => chooseLineMode(checked ? 'gpu' : 'mean'), + infoTooltip: t.perGpuHelp, + }, + ]; + if (hasPools) { + switches.push({ + id: 'power-timeline-pools', + label: t.pools, + checked: lineMode === 'pool', + onCheckedChange: (checked) => chooseLineMode(checked ? 'pool' : 'mean'), + infoTooltip: t.poolsHelp, + }); + } + switches.push({ + id: 'power-timeline-utility', + label: t.utilityLines, + checked: showUtility, + onCheckedChange: (checked) => { + setShowUtility(checked); + track('inference_power_timeline_utility_toggled', { enabled: checked }); + }, + infoTooltip: t.utilityHelp, + }); + + const legendElement = ( + { + setIsLegendExpanded(expanded); + track('latency_legend_expanded', { expanded }); + }} + onItemHover={(id) => setHighlight(id)} + onItemHoverEnd={() => setHighlight(null)} + hideAtomFootnote + switches={switches} + /> + ); + + const runInfos = useMemo( + () => + fetchedRequests + .map((request) => responses.get(request.runId)?.runInfo) + .filter((info): info is NonNullable => Boolean(info)), + [fetchedRequests, responses], + ); + + const toolbar = ( +
    + {t.timeAxis} + + value={xMode} + ariaLabel={t.timeAxis} + role="group" + options={[ + { value: 'wall', label: t.wall, testId: 'power-timeline-axis-wall' }, + { value: 'elapsed', label: t.elapsed, testId: 'power-timeline-axis-elapsed' }, + ]} + onValueChange={(mode) => { + setXModeChoice(mode); + track('inference_power_timeline_axis_changed', { mode }); + }} + /> +
    + ); + + let emptyMessage: string | null = null; + if (loadingRuns > 0) emptyMessage = t.loading(loadingRuns); + else if (!hasAnyArtifact) emptyMessage = t.noArtifacts; + else if (visibleTraces.length === 0) emptyMessage = t.noTraces; + + return ( +
    + + key={`${chartId}-${xMode}-${lineMode}`} + chartId={chartId} + data={samples} + height={CHART_HEIGHT} + margin={MARGIN} + watermark={overlayData && overlayPoints.length > 0 ? 'unofficial' : 'logo'} + testId="power-timeline-chart-svg" + grabCursor + instructions={t.instructions} + xScale={ + xMode === 'wall' + ? { type: 'time', domain: [new Date(xDomain[0]), new Date(xDomain[1])] } + : { type: 'linear', domain: xDomain } + } + yScale={{ type: 'linear', domain: yDomain, nice: true }} + xAxis={{ + label: xMode === 'wall' ? t.xWall : t.xElapsed, + tickValues: (scale) => { + const timeScale = scale as + | d3.ScaleTime + | d3.ScaleLinear; + const [left, right] = timeScale.range(); + return timeScale.ticks(Math.max(2, Math.min(10, Math.floor((right - left) / 80)))); + }, + tickFormat: xTickFormat, + }} + yAxis={{ label: lineMode === 'pool' ? t.yPool : yLabel, tickCount: 8 }} + layers={layers} + displayIdentity={activeHighlight ?? ''} + onDisplayUpdate={onDisplayUpdate} + zoom={{ + enabled: true, + axes: 'x', + scaleExtent: [1, 60], + resetEventName: `power_timeline_zoom_reset_${chartId}`, + }} + tooltip={{ + rulerType: 'crosshair', + content: tooltipContent, + getRulerX: (sample, xScale) => (xScale as AnyContinuousScale)(sample.x), + getRulerY: (sample, yScale) => yScale(sample.y), + onHoverStart: (selection) => { + selection.attr('r', 5).attr('stroke', 'white').attr('stroke-width', 1); + }, + onHoverEnd: (selection) => { + selection.attr('r', 2).attr('stroke', 'none'); + }, + attachToLayer: 2, + }} + legendElement={legendElement} + caption={ + <> + {caption} + {toolbar} + + } + noDataOverlay={ + emptyMessage ? ( +
    + {emptyMessage} +
    + ) : undefined + } + /> +
    + {focusedTrace && ( +

    + + {t.focused( + `${hardwareLabel(focusedTrace.point)} · ${traceConfigLabel(focusedTrace.point)}`, + )} + + +

    + )} + {errors.map(({ request, error }) => ( +

    + {t.loadError(request.runId, error.message)} +

    + ))} + {droppedRuns > 0 &&

    {t.droppedRuns(droppedRuns)}

    } + {loadingRuns === 0 && missing.length > 0 && traces.length > 0 && ( +

    + {t.missing(missing.length, traces.length + missing.length)}{' '} + {describeMissing(missing, t, hardwareLabel)} +

    + )} + {runInfos.length > 0 && ( +

    + {t.telemetry}:{' '} + {runInfos.map((info, index) => ( + + {index > 0 && ' · '} + + {`run ${info.id}`} + + {info.createdAt ? ` (${formatUtcDate(new Date(info.createdAt))})` : ''} + + ))} +

    + )} +

    {t.method}

    + {hasPools &&

    {t.methodPools}

    } +
    +
    + ); +} diff --git a/packages/app/src/components/inference/ui/ScatterGraph.test-harness.tsx b/packages/app/src/components/inference/ui/ScatterGraph.test-harness.tsx index 2cebdba58..5e72fb046 100644 --- a/packages/app/src/components/inference/ui/ScatterGraph.test-harness.tsx +++ b/packages/app/src/components/inference/ui/ScatterGraph.test-harness.tsx @@ -80,6 +80,10 @@ class MockResizeObserver { unobserve() {} disconnect() {} } +const originalGetComputedTextLength = Object.getOwnPropertyDescriptor( + SVGElement.prototype, + 'getComputedTextLength', +); const originalGetBBox = Object.getOwnPropertyDescriptor(SVGElement.prototype, 'getBBox'); export const point = ( @@ -220,6 +224,12 @@ export const rebuildCount = () => vi.mocked(setupChartStructure).mock.calls.leng beforeEach(() => { globalThis.__scatterPathnameState.value = '/inference'; vi.stubGlobal('ResizeObserver', MockResizeObserver); + Object.defineProperty(SVGElement.prototype, 'getComputedTextLength', { + configurable: true, + value(this: SVGElement) { + return (this.textContent?.length ?? 0) * 7; + }, + }); Object.defineProperty(SVGElement.prototype, 'getBBox', { configurable: true, value: () => @@ -259,6 +269,15 @@ beforeEach(() => { afterEach(() => { vi.unstubAllGlobals(); vi.restoreAllMocks(); + if (originalGetComputedTextLength) { + Object.defineProperty( + SVGElement.prototype, + 'getComputedTextLength', + originalGetComputedTextLength, + ); + } else { + Reflect.deleteProperty(SVGElement.prototype, 'getComputedTextLength'); + } if (originalGetBBox) { Object.defineProperty(SVGElement.prototype, 'getBBox', originalGetBBox); } else { diff --git a/packages/app/src/components/inference/ui/ScatterGraph.tsx b/packages/app/src/components/inference/ui/ScatterGraph.tsx index fc860d3ad..318e2dce8 100644 --- a/packages/app/src/components/inference/ui/ScatterGraph.tsx +++ b/packages/app/src/components/inference/ui/ScatterGraph.tsx @@ -1,10 +1,13 @@ 'use client'; +import { useFeatureGate } from '@/lib/use-feature-gate'; +import { getMeasuredMetricConfig } from '@/components/inference/measured-metric-config'; import { track } from '@/lib/analytics'; import { isPersistedBenchmarkId } from '@/lib/benchmark-id'; import { useEphemeralUrlState } from '@/hooks/useUrlState'; import { rememberChartStateInUrl } from '@/lib/url-state'; import * as d3 from 'd3'; +import { CHART_TYPE } from '@/lib/d3-chart/typography'; import dynamic from 'next/dynamic'; import React, { useCallback, useEffect, useLayoutEffect, useMemo, useRef, useState } from 'react'; @@ -16,6 +19,7 @@ import { useInferenceDisplay, useInferenceFilters, } from '@/components/inference/InferenceContext'; +import { usePerfRulerStore } from '@/components/inference/perf-ruler-store'; import { useTraceAvailability } from '@/hooks/api/use-trace-availability'; import { useLogAvailability } from '@/hooks/api/use-log-availability'; import { computeToggle } from '@/hooks/useTogglableSet'; @@ -47,7 +51,13 @@ import { matchKnownConfigIssues, pointMatchesIssue } from '@/lib/known-issues'; import { useLocale } from '@/lib/use-locale'; import { getLineLabelVendorIcon } from '@/lib/vendor-logos'; import { formatNumber, getDisplayLabel, updateRepoUrl } from '@/lib/utils'; -import { getInferenceHardwareConfig, getInferenceRunLabel } from '@/lib/inference-labels'; +import { + getInferenceHardwareConfig, + getInferenceRunLabel, + getOverlayLineLabel, + OVERLAY_LABEL_MARKER, + overlayRunTag, +} from '@/lib/inference-labels'; import { D3Chart } from '@/lib/d3-chart/D3Chart'; import type { CustomLayerConfig, @@ -76,6 +86,7 @@ import { renderPerfRulers, type PerfRulerEndInput, type PerfRulerGeometry, + type PerfRulerMeasurement, type PerfRulerRenderEntry, type PerfRulerState, } from '@/lib/d3-chart/layers/perf-ruler'; @@ -111,6 +122,7 @@ import { chartFrontier, upperPowerEnvelope, isPowerCurveMetric, + isPowerGaugeSeries, isMeasuredPowerCurveMetric, } from '@/components/inference/utils/powerCurves'; import type { @@ -118,17 +130,38 @@ import type { ClippedInferenceData, InferenceData, ScatterGraphProps, + PowerVariant, } from '@/components/inference/types'; import { generateOverlayTooltipContent, generateTooltipContent, } from '@/components/inference/utils/tooltipUtils'; +import { + POWER_TIMELINE_METRIC_KEY, + requestPowerTraceFocus, + traceKeyForPoint, +} from '@/components/inference/utils/powerTimeline'; import { QuickFiltersDialog } from '@/components/inference/ui/QuickFiltersDialog'; import { ScatterEmptyState } from '@/components/inference/ui/ScatterEmptyState'; import { scatterPointConfigId, scatterPointJoinId, + parseScatterSeriesKey, + scatterSeriesKey, } from '@/components/inference/utils/point-identity'; +import { + flatSeriesValue, + inferPowerCompare, + lineLabelHardwareKey, + lineLabelSeriesId, + metricPlotsWatts, + powerCompareBase, + powerLineLabelSuffix, + powerVariantDash, + powerVariantId, + powerVariantLabel, + powerVariantsInData, +} from '@/components/inference/utils/power-compare'; import LegendPointsDialog from '@/components/inference/ui/LegendPointsDialog'; import { renderOffloadHalo } from '@/components/inference/utils/offload-halo'; import { renderLegacyPowerRing } from '@/components/inference/utils/legacy-power-marker'; @@ -163,6 +196,14 @@ import { fitContinuationLabelBaseline, } from '@/components/inference/utils/overflowContinuations'; +const PowerTelemetryDialog = dynamic( + () => + import('@/components/inference/power-telemetry-dialog').then( + (module) => module.PowerTelemetryDialog, + ), + { ssr: false }, +); + const FixedSequenceLogDialog = dynamic( () => import('@/components/inference/log-viewer/fixed-sequence-log-dialog').then( @@ -221,6 +262,32 @@ const optimalPointKey = (d: InferenceData): string => const EMPTY_OVERLAY_DATA: InferenceData[] = []; const EMPTY_CLIPPED_DATA: ClippedInferenceData[] = []; +/** + * Legend ids of the comparison-series rows (`i_pcompare`), distinct from + * hardware keys so the shared hover / toggle handlers can tell them apart. + */ +const POWER_VARIANT_LEGEND_PREFIX = 'power-variant:'; +/** Comparison clones sit behind the base series they annotate. */ +const POWER_VARIANT_POINT_OPACITY = 0.6; +const pointOpacityForVariant = (d: InferenceData): number => + d.powerVariant ? POWER_VARIANT_POINT_OPACITY : 1; +/** Dash for a series key's variant id (`parseScatterSeriesKey().variant`). */ +const powerVariantDashById = (variantId: string | null | undefined): string => + variantId ? (VARIANT_DASH_BY_ID.get(variantId) ?? '') : ''; +const VARIANT_DASH_BY_ID = new Map( + ( + [ + ['basis', 'gpu-measured'], + ['basis', 'gpu-provisioned'], + ['basis', 'utility-provisioned'], + ['basis', 'utility-modeled'], + ['role', 'all'], + ['role', 'prefill'], + ['role', 'decode'], + ] as const + ).map(([kind, id]) => [id, powerVariantDash({ kind, id } as PowerVariant)]), +); + const LINE_LABEL_RAISE = ['.line-label'] as const; /** Decorations sit above the visible shape, which a precision toggle may replace. */ const POINT_DECORATION_RAISE = ['.offload-halo', '.legacy-power-ring'] as const; @@ -534,6 +601,7 @@ const ScatterGraph = React.memo( setQuickFilterDeployment, setQuickFilterSpec, setQuickFilterPower, + setSelectedYAxisMetric, } = useInferenceActions(); const paretoDirection = chartDefinition[`${selectedYAxisMetric}_roofline`] as | ParetoDirection @@ -552,15 +620,31 @@ const ScatterGraph = React.memo( const groups = groupPointsByDate(points); if (showPowerEnvelope) { for (const [date, samples] of groups) { - groups.set(date, upperPowerEnvelope(samples, chartDefinition.chartType !== 'e2e')); + groups.set( + date, + upperPowerEnvelope( + samples, + chartDefinition.chartType !== 'e2e', + isPowerGaugeSeries(selectedYAxisMetric, samples[0]), + ), + ); } } return groups; }, - [showPowerEnvelope, chartDefinition.chartType], + [showPowerEnvelope, chartDefinition.chartType, selectedYAxisMetric], ); const locale = useLocale(); + const featureGateUnlocked = useFeatureGate(); + const showPowerTelemetry = + featureGateUnlocked || getMeasuredMetricConfig(selectedYAxisMetric) !== undefined; const legendT = SCATTER_STRINGS[locale]; + // Comparison series (`i_pcompare`) switched off from the legend. Chart-local, + // like Optimal Only's point set: the URL carries the comparison, not which + // of its rows a reader hid while looking. + const [hiddenPowerVariants, setHiddenPowerVariants] = useState>( + () => new Set(), + ); const ephemeralUrlState = useEphemeralUrlState(); const costLimit = chartDefinition.y_cost_limit ?? 0; const latencyLimit = chartDefinition.y_latency_limit ?? 0; @@ -801,7 +885,7 @@ const ScatterGraph = React.memo( () => data.reduce( (acc, point) => { - const key = `${point.hwKey}_${point.precision}`; + const key = scatterSeriesKey(point); if (!acc[key]) acc[key] = []; acc[key].push(point); return acc; @@ -938,7 +1022,7 @@ const ScatterGraph = React.memo( } const buckets = new Map(); const getBucket = (point: InferenceData) => { - const key = `${point.hwKey}|${point.precision}|${point.date}`; + const key = `${scatterSeriesKey(point)}|${point.date}`; let bucket = buckets.get(key); if (!bucket) { bucket = { @@ -987,7 +1071,7 @@ const ScatterGraph = React.memo( const buckets = new Map(); const getBucket = (point: InferenceData) => { const runIndex = overlayRunIndex(point.run_url ?? null, runIndexByUrl); - const key = `${point.hwKey}|${point.precision}|${point.date}|run${runIndex}`; + const key = `${scatterSeriesKey(point)}|${point.date}|run${runIndex}`; let bucket = buckets.get(key); if (!bucket) { bucket = { @@ -1092,6 +1176,8 @@ const ScatterGraph = React.memo( interface Entry { hwKey: string; runIndex: number; + /** Comparison variant id for boundary / role clones, null for the run's base series. */ + variant: string | null; points: InferenceData[]; } if (processedOverlayData.length === 0) return {} as Record; @@ -1100,8 +1186,15 @@ const ScatterGraph = React.memo( const grouped = processedOverlayData.reduce( (acc, p) => { const runIndex = overlayRunIndex(p.run_url ?? null, runIndexByUrl); - const key = `${p.hwKey}_${p.precision}_run${runIndex}`; - if (!acc[key]) acc[key] = { hwKey: String(p.hwKey), runIndex, points: [] }; + const key = `${scatterSeriesKey(p)}_run${runIndex}`; + if (!acc[key]) { + acc[key] = { + hwKey: String(p.hwKey), + runIndex, + variant: p.powerVariant?.id ?? null, + points: [], + }; + } acc[key].points.push(p); return acc; }, @@ -1180,9 +1273,11 @@ const ScatterGraph = React.memo( // its X marker sitting on the dashed roofline and read as a pareto point. const isOverlayPointVisible = useCallback( (d: InferenceData) => + !hiddenPowerVariants.has(powerVariantId(d.powerVariant)) && (!hideNonOptimal || overlayOptimalPoints.has(d)) && (!showPowerEnvelope || showAllMeasurements || overlayEnvelopePoints.has(d)), [ + hiddenPowerVariants, hideNonOptimal, overlayOptimalPoints, showPowerEnvelope, @@ -1216,6 +1311,45 @@ const ScatterGraph = React.memo( ); const { data: persistedLogAvailability } = useLogAvailability(persistedPointIds); const [fixedLogPointId, setFixedLogPointId] = useState(null); + const [powerTelemetryPoint, setPowerTelemetryPoint] = useState(null); + + // "View power trace" on a pinned tooltip (official or overlay point): the + // same-tab click stays in-page — remember which trace to emphasise, switch + // the metric to the Timeline display, and let the anchor's href keep + // serving open-in-new-tab. Listeners are attached per pin because the + // tooltip HTML is replaced on every pin. + const attachPowerTraceAction = useCallback( + (tooltipEl: HTMLElement, d: InferenceData, overlay: boolean) => { + const action = tooltipEl.querySelector('[data-action="view-power-trace"]'); + const traceKey = traceKeyForPoint(d); + if (!action || !traceKey) return; + action.addEventListener('click', (actionEvent) => { + actionEvent.stopPropagation(); + // Modifier / auxiliary clicks keep the anchor's own behaviour: the + // href opens this chart's timeline in a new tab or window. + const mouse = actionEvent as MouseEvent; + if ( + mouse.button !== 0 || + mouse.metaKey || + mouse.ctrlKey || + mouse.shiftKey || + mouse.altKey + ) { + return; + } + actionEvent.preventDefault(); + requestPowerTraceFocus(traceKey); + chartRef.current?.dismissTooltip(); + setSelectedYAxisMetric(POWER_TIMELINE_METRIC_KEY); + track('inference_power_trace_opened', { + hwKey: String(d.hwKey), + conc: d.conc, + overlay, + }); + }); + }, + [setSelectedYAxisMetric], + ); // --- Legend points table (per-series drill-down opened from the legend) --- const [pointsTableTarget, setPointsTableTarget] = useState(null); @@ -1251,6 +1385,7 @@ const ScatterGraph = React.memo( const pts = pointsData.filter( (p) => p.hwKey === hwKey && + !p.powerVariant && selectedPrecisions.includes(p.precision) && (!hideNonOptimal || optimalPointKeys.has(optimalPointKey(p))), ); @@ -1266,6 +1401,7 @@ const ScatterGraph = React.memo( const pts = processedOverlayData.filter( (p) => overlayRunIndex(p.run_url ?? null, runIndexByUrl) === runIndex && + !p.powerVariant && activeOverlayHwTypes.has(p.hwKey as string) && (!hideNonOptimal || overlayOptimalPoints.has(p)), ); @@ -1486,6 +1622,7 @@ const ScatterGraph = React.memo( (d: InferenceData) => effectiveActiveHwTypes.has(d.hwKey as string) && selectedPrecisions.includes(d.precision) && + !hiddenPowerVariants.has(powerVariantId(d.powerVariant)) && (!hideNonOptimal || optimalPointKeys.has(optimalPointKey(d))) && (!showPowerEnvelope || showAllMeasurements || @@ -1493,6 +1630,7 @@ const ScatterGraph = React.memo( [ effectiveActiveHwTypes, selectedPrecisions, + hiddenPowerVariants, hideNonOptimal, optimalPointKeys, showPowerEnvelope, @@ -1669,12 +1807,69 @@ const ScatterGraph = React.memo( getCssColor, ]); + // The comparison in effect and the base series' identity under it. The + // base is the selected metric's own series; deriving it from which variant + // no official point carries breaks when only an overlay carries the + // comparison, and line labels need the same answer as the legend rows. + const powerCompareMode = useMemo(() => { + const official = inferPowerCompare(pointsData); + return official === 'none' ? inferPowerCompare(processedOverlayData) : official; + }, [pointsData, processedOverlayData]); + const powerCompareBaseId = useMemo( + () => powerVariantId(powerCompareBase(selectedYAxisMetric, powerCompareMode)), + [selectedYAxisMetric, powerCompareMode], + ); + + // One legend row per comparison series present (base first). Rows toggle + // chart-local visibility and hover-highlight that series across hardware. + const powerVariantLegendItems = useMemo(() => { + const allPoints = [...pointsData, ...processedOverlayData]; + const variants = powerVariantsInData(allPoints, selectedYAxisMetric); + const baseId = powerCompareBaseId; + return variants.map((variant) => { + const id = powerVariantId(variant); + const legendId = `${POWER_VARIANT_LEGEND_PREFIX}${id}`; + const isBase = id === baseId; + return { + name: legendId, + hw: legendId, + label: powerVariantLabel(variant, locale), + color: 'var(--foreground)', + // The base series is solid, like its points; siblings carry their dash. + lineDasharray: isBase ? '1 0' : powerVariantDash(variant) || '1 0', + isActive: !hiddenPowerVariants.has(isBase ? '' : id), + isRemovable: false, + onClick: () => { + const key = isBase ? '' : id; + setHiddenPowerVariants((prev) => { + const next = new Set(prev); + if (next.has(key)) next.delete(key); + else next.add(key); + return next; + }); + track('inference_power_compare_series_toggled', { + series: id, + visible: hiddenPowerVariants.has(key), + }); + }, + }; + }); + }, [ + pointsData, + processedOverlayData, + selectedYAxisMetric, + powerCompareBaseId, + locale, + hiddenPowerVariants, + ]); + const powerTierCounts = useMemo(() => { - const officialTotal = pointsData.filter((point) => - selectedPrecisions.includes(point.precision), + // Comparison clones re-plot the same measurements; count each once. + const officialTotal = pointsData.filter( + (point) => !point.powerVariant && selectedPrecisions.includes(point.precision), ); - const overlayTotal = processedOverlayData.filter((point) => - selectedPrecisions.includes(point.precision), + const overlayTotal = processedOverlayData.filter( + (point) => !point.powerVariant && selectedPrecisions.includes(point.precision), ); const officialVisible = officialTotal.filter(isPointVisible); const overlayVisible = overlayTotal.filter( @@ -1699,9 +1894,13 @@ const ScatterGraph = React.memo( const hw = el.dataset.hwKey; const prec = el.dataset.precision; if (hw === null || hw === undefined || prec === null || prec === undefined) return false; - return effectiveActiveHwTypes.has(hw) && selectedPrecisions.includes(prec); + return ( + effectiveActiveHwTypes.has(hw) && + selectedPrecisions.includes(prec) && + !hiddenPowerVariants.has(el.dataset.powerVariant ?? '') + ); }, - [effectiveActiveHwTypes, selectedPrecisions], + [effectiveActiveHwTypes, selectedPrecisions, hiddenPowerVariants], ); // --- Interaction state ref --- @@ -1718,6 +1917,7 @@ const ScatterGraph = React.memo( isPointVisible, isOverlayPointVisible, effectiveActiveHwTypes, + hiddenPowerVariants, selectedPrecisions, activeOverlayHwTypes, getCssColor, @@ -1725,11 +1925,13 @@ const ScatterGraph = React.memo( knownIssueAnnotations, traceAvailability, logAvailability: persistedLogAvailability, + showPowerTelemetry, }); interactionRef.current = { isPointVisible, isOverlayPointVisible, effectiveActiveHwTypes, + hiddenPowerVariants, selectedPrecisions, activeOverlayHwTypes, getCssColor, @@ -1737,6 +1939,7 @@ const ScatterGraph = React.memo( knownIssueAnnotations, traceAvailability, logAvailability: persistedLogAvailability, + showPowerTelemetry, }; // --- Perf ruler (opt-in: click two curves, drag the ruler to any iso-x) --- @@ -1748,17 +1951,39 @@ const ScatterGraph = React.memo( // the curves' rendered paths at the iso-x — neither end needs to be a // data point. Multiple rulers accumulate (capped in the pure module); // completing one immediately allows starting the next. - const [preferPerfRulerMode, setPerfRulerMode] = useState(false); + // + // The primary chart's rulers live in the InferenceProvider store so they + // ride along in share links (`i_rulers`) and survive a remount (table + // view toggle). Every other instance — the replay chart, which draws the + // same curve classes, and harnesses mounted without the provider — keeps + // component-local state. Both paths share one `[state, setState]` pair + // below, so the reducers, refs, and draw passes are path-agnostic. + const perfRulerStore = usePerfRulerStore(); + const persistedRulers = perfRulerStore?.chartId === chartId ? perfRulerStore : undefined; + // Rulers only render while the mode is on (and the mode-off effect below + // clears them), so restored share-link rulers — pending or already + // committed by a previous mount — switch the mode on for this instance. + const [preferPerfRulerMode, setPerfRulerMode] = useState( + () => + persistedRulers !== undefined && + (persistedRulers.pending !== null || persistedRulers.state.rulers.length > 0), + ); const perfRulerMode = preferPerfRulerMode && (!showPowerEnvelope || isMeasuredPowerAxis); - const [perfRulerState, setPerfRulerState] = useState(EMPTY_PERF_RULER_STATE); + const [localPerfRulerState, setLocalPerfRulerState] = + useState(EMPTY_PERF_RULER_STATE); + const perfRulerState = persistedRulers ? persistedRulers.state : localPerfRulerState; + const setPerfRulerState = persistedRulers ? persistedRulers.setState : setLocalPerfRulerState; // Changing the x- or y-axis metric (including the x percentile, which // `x_scale_field` encodes) clears every ruler: the curves are redrawn // in different units, so a ruler that persisted would measure a ratio // the user never placed. Runs before the draw pass so no stale ruler - // ever paints over the new curves. + // ever paints over the new curves. Render-time adjustment is only legal + // for this component's own state, so the hook targets the local state; + // the store applies the same reset to persisted rulers inside the + // provider (see usePerfRulerStoreValue). usePerfRulerAxisReset( perfRulerAxisMetricKey(chartDefinition.x_scale_field, selectedYAxisMetric), - setPerfRulerState, + setLocalPerfRulerState, ); // Draw passes read mode/state through refs so toggling off clears the // rulers in the same pre-paint layout pass — lines/labels must never @@ -1824,6 +2049,51 @@ const ScatterGraph = React.memo( [], ); + // Share-link rulers commit only once BOTH curve paths are in the DOM — + // otherwise the prune pass would eat them before their data (i_gpus, + // comparison dates, overlay runs) has arrived. Hidden curves (opacity 0) + // count as present, like for prune. The iso-x is clamped to the pair's + // overlap through the drawn paths, so a rounded or since-shifted iso-x + // still renders; a pair with disjoint spans can never be measured on + // these axes and is dropped. This runs from the draw pass rather than a + // React effect: the chart first draws in a D3Chart-local re-render + // (dimensions are measured after mount), which re-renders nothing here, + // so an effect keyed on our props could miss the first draw and leave + // resolvable rulers pending for the rest of the session. The store is + // read through a ref for the same reason the draw passes read the ruler + // state through refs. Nothing commits while the mode is off (forced off + // by the power envelope, or switched off by the user) — the mode-off + // effect discards pending rulers, and the analytics event must not + // report a restore nobody saw. Draw passes can repeat before React has + // applied a commit, so the pending list handed over is remembered by + // identity and skipped until the store replaces it. + const persistedRulersRef = useRef(persistedRulers); + persistedRulersRef.current = persistedRulers; + const committedPendingRef = useRef(null); + const commitPendingPerfRulers = useCallback( + (zoomGroup: d3.Selection) => { + const store = persistedRulersRef.current; + const pending = store?.pending ?? null; + if (!store || !pending || !perfRulerModeRef.current) return; + if (committedPendingRef.current === pending) return; + const curveExists = (cls: string) => !zoomGroup.select(`.${CSS.escape(cls)}`).empty(); + const resolved: PerfRulerMeasurement[] = []; + const remaining: PerfRulerMeasurement[] = []; + for (const ruler of pending) { + if (!curveExists(ruler.curveA) || !curveExists(ruler.curveB)) { + remaining.push(ruler); + continue; + } + const isoX = clampPerfRulerIsoXToOverlap(ruler.curveA, ruler.curveB, ruler.isoX); + if (isoX !== null) resolved.push({ ...ruler, isoX }); + } + if (remaining.length === pending.length) return; + committedPendingRef.current = pending; + store.commitPending(resolved, remaining.length > 0 ? remaining : null); + }, + [clampPerfRulerIsoXToOverlap], + ); + // Curve click (widened hit strokes): iso-x is the click's x pixel // through the CURRENT rendered x scale, stored in data space. const handlePerfRulerCurveClick = useCallback( @@ -1852,7 +2122,7 @@ const ScatterGraph = React.memo( (point: InferenceData, source: 'official' | 'overlay') => { const ctx = perfRulerDrawCtxRef.current; if (!ctx) return; - const series = `${String(point.hwKey)}_${point.precision}`; + const series = scatterSeriesKey(point); const base = source === 'overlay' ? `overlay-roofline-${series}_run${overlayRunIndex(point.run_url ?? null, runIndexByUrl)}` @@ -1882,8 +2152,12 @@ const ScatterGraph = React.memo( // the switch handler also clears synchronously, this covers // programmatic mode changes). `clearPerfRulers` bails out with the same // reference when there is nothing to clear. + // Share-link rulers still waiting for their curves go too — the user + // switched the tool off, so nothing should surface later. useEffect(() => { - if (!perfRulerMode) setPerfRulerState(clearPerfRulers); + if (perfRulerMode) return; + setPerfRulerState(clearPerfRulers); + persistedRulers?.discardPending(); }, [perfRulerMode]); // Invisible widened hit strokes over every rendered roofline path @@ -2033,6 +2307,7 @@ const ScatterGraph = React.memo( ) => { perfRulerDrawCtxRef.current = { zoomGroup, xScale, yScale, width, height }; syncPerfRulerHitPaths(zoomGroup); + commitPendingPerfRulers(zoomGroup); const state = perfRulerStateRef.current; const entries: PerfRulerRenderEntry[] = []; if (perfRulerModeRef.current && state.rulers.length > 0) { @@ -2086,7 +2361,7 @@ const ScatterGraph = React.memo( ); if (!dragHandles.empty()) dragHandles.call(perfRulerDrag); }, - [syncPerfRulerHitPaths, perfRulerDrag], + [syncPerfRulerHitPaths, commitPendingPerfRulers, perfRulerDrag], ); drawPerfRulerRef.current = drawPerfRuler; @@ -2120,16 +2395,31 @@ const ScatterGraph = React.memo( const svg = chartRef.current?.getSvgElement?.(); if (!svg) return; const root = d3.select(svg); + // A comparison-series legend row highlights that boundary / role + // across every hardware instead of one hardware across series. Base + // points and rooflines carry no variant (only siblings are cloned), + // so the empty id maps back to the base legend row's id. + const variantId = hwKey.startsWith(POWER_VARIANT_LEGEND_PREFIX) + ? hwKey.slice(POWER_VARIANT_LEGEND_PREFIX.length) + : null; + const matchesPoint = (d: InferenceData) => + variantId === null + ? String(d.hwKey) === hwKey + : (powerVariantId(d.powerVariant) || powerCompareBaseId) === variantId; root .selectAll('.dot-group') .style('opacity', (d) => - isPointVisible(d) ? (String(d.hwKey) === hwKey ? 1 : 0.15) : 0, + isPointVisible(d) ? (matchesPoint(d) ? pointOpacityForVariant(d) : 0.15) : 0, ); root .selectAll('.roofline-path, .official-overflow-continuation') .style('opacity', function () { if (!isRooflineVisible(this)) return 0; - return this.dataset.hwKey === hwKey ? null : '0.15'; + const matches = + variantId === null + ? this.dataset.hwKey === hwKey + : (this.dataset.powerVariant || powerCompareBaseId) === variantId; + return matches ? null : '0.15'; }); root .selectAll('.parallelism-label, .line-label') @@ -2137,7 +2427,7 @@ const ScatterGraph = React.memo( return labelOpacityForHover((this as SVGGElement).dataset, hwKey); }); }, - [isPointVisible, isRooflineVisible], + [isPointVisible, isRooflineVisible, powerCompareBaseId], ); const handleLegendHoverEnd = useCallback(() => { @@ -2146,7 +2436,7 @@ const ScatterGraph = React.memo( const root = d3.select(svg); root .selectAll('.dot-group') - .style('opacity', (d) => (isPointVisible(d) ? 1 : 0)); + .style('opacity', (d) => (isPointVisible(d) ? pointOpacityForVariant(d) : 0)); root .selectAll('.roofline-path, .official-overflow-continuation') .style('opacity', function () { @@ -2159,9 +2449,16 @@ const ScatterGraph = React.memo( (this as SVGGElement).dataset, effectiveActiveHwTypes, selectedPrecisions, + activeOverlayHwTypes, ); }); - }, [isPointVisible, isRooflineVisible, effectiveActiveHwTypes, selectedPrecisions]); + }, [ + isPointVisible, + isRooflineVisible, + effectiveActiveHwTypes, + selectedPrecisions, + activeOverlayHwTypes, + ]); // --- Zoom config --- const eventPrefix = chartDefinition.chartType === 'e2e' ? 'latency' : 'interactivity'; @@ -2230,6 +2527,7 @@ const ScatterGraph = React.memo( yLabel, selectedYAxisMetric, hardwareConfig, + showPowerTelemetry: interactionRef.current.showPowerTelemetry, runUrl: d.run_url ? updateRepoUrl(d.run_url) : undefined, hasTrace: d.benchmark_type === 'agentic_traces' && isPersistedBenchmarkId(d.id) @@ -2291,6 +2589,15 @@ const ScatterGraph = React.memo( }); }); } + const powerBtn = tooltipEl.querySelector('[data-action="view-power-telemetry"]'); + if (powerBtn && isPersistedBenchmarkId(d.id)) { + powerBtn.addEventListener('click', (event) => { + event.stopPropagation(); + setPowerTelemetryPoint(d); + chartRef.current?.dismissTooltip(); + track('inference_power_telemetry_opened', { id: d.id, hwKey: d.hwKey, conc: d.conc }); + }); + } const logsBtn = tooltipEl.querySelector('[data-action="view-logs"]'); if (logsBtn && typeof d.id === 'number') { logsBtn.addEventListener('click', (btnEvent) => { @@ -2308,10 +2615,12 @@ const ScatterGraph = React.memo( }); }); } + attachPowerTraceAction(tooltipEl, d, false); }, attachToLayer: 1, // scatter layer is index 1 (after rooflines at 0) }), [ + attachPowerTraceAction, xLabel, yLabel, selectedYAxisMetric, @@ -2325,6 +2634,31 @@ const ScatterGraph = React.memo( // --- Layers --- const layers = useMemo((): LayerConfig[] => { + // Line-label identity of one drawn series under a power comparison + // (`i_pcompare`): the base series keeps the hardware key, so pinned + // anchors and hover hooks keep working; a sibling is `::`. + const wattsAxis = metricPlotsWatts(selectedYAxisMetric); + const lineLabelIdentity = (hw: string, points: readonly InferenceData[]) => { + const variant = points[0]?.powerVariant; + const variantId = powerVariantId(variant); + const isBase = !variant || variantId === powerCompareBaseId; + return { variant, variantId, isBase, seriesId: lineLabelSeriesId(hw, variant, isBase) }; + }; + // A sibling's label says which series it is; a flat provisioned boundary + // (TDP, all-in) on a watts axis also states its value. + const lineLabelSuffix = ( + identity: ReturnType, + points: readonly InferenceData[], + ) => + powerLineLabelSuffix(identity.variant, { + isBase: identity.isBase, + locale, + flatWatts: + !identity.isBase && wattsAxis && identity.variant?.kind === 'basis' + ? flatSeriesValue(points.map((point) => point.y)) + : null, + }); + // ── Layer 0: Rooflines + gradient labels (custom) ── const rooflineLayer: CustomLayerConfig = { type: 'custom', @@ -2362,6 +2696,8 @@ const ScatterGraph = React.memo( key: string; hw: string; precision: string; + /** Comparison variant id (`i_pcompare`), '' for the base series. */ + variant: string; points: InferenceData[]; stroke: string; visible: boolean; @@ -2370,10 +2706,11 @@ const ScatterGraph = React.memo( const activeGradientIds = new Set(); Object.entries(displayedRooflines).forEach(([key, pts]) => { - const hw = key.split('_').slice(0, -1).join('_'); - const precision = key.split('_').pop()!; + const { hw, precision, variant } = parseScatterSeriesKey(key); const visible = - ir.effectiveActiveHwTypes.has(hw) && ir.selectedPrecisions.includes(precision); + ir.effectiveActiveHwTypes.has(hw) && + ir.selectedPrecisions.includes(precision) && + !ir.hiddenPowerVariants.has(variant ?? ''); const baseStroke = ir.getCssColor(ir.resolveColor(hw)); // Split into per-date sub-paths so the line never crosses dates. @@ -2417,6 +2754,7 @@ const ScatterGraph = React.memo( key: entryKey, hw, precision, + variant: variant ?? '', points: datePoints, stroke, visible, @@ -2444,9 +2782,12 @@ const ScatterGraph = React.memo( .attr('data-curve-kind', showPowerEnvelope ? 'power-envelope' : 'pareto') .attr('data-hw-key', (d) => d.hw) .attr('data-precision', (d) => d.precision) + .attr('data-power-variant', (d) => d.variant || null) .attr('fill', 'none') .attr('stroke', (d) => d.stroke) .attr('stroke-width', 2.5) + // Comparison siblings share the hardware colour; the dash tells them apart. + .attr('stroke-dasharray', (d) => powerVariantDashById(d.variant) || null) .attr('d', (d) => lineGen(d.points)) .style('transition', 'opacity 150ms ease') .style('opacity', (d) => (d.visible ? 1 : 0)); @@ -2467,10 +2808,11 @@ const ScatterGraph = React.memo( if (showGradientLabels) { Object.entries(allPointLabelsByKey).forEach(([key, pointLabels]) => { if (pointLabels.length < 2) return; - const hw = key.split('_').slice(0, -1).join('_'); - const precision = key.split('_').pop()!; + const { hw, precision, variant } = parseScatterSeriesKey(key); const visible = - ir.effectiveActiveHwTypes.has(hw) && ir.selectedPrecisions.includes(precision); + ir.effectiveActiveHwTypes.has(hw) && + ir.selectedPrecisions.includes(precision) && + !ir.hiddenPowerVariants.has(variant ?? ''); const segments: { label: string; color: string; points: InferenceData[] }[] = []; let cur = { @@ -2568,12 +2910,22 @@ const ScatterGraph = React.memo( // ── Line labels (run name along each roofline) ── let lineLabels: LineLabelPlacement[] = []; + // Comparison variant and label suffix per label key, for the text + // segments and the `data-power-variant` hook on each pill. + const lineLabelMeta = new Map< + string, + { variantId: string; suffix: string; runTag: string } + >(); if (showLineLabels) { const multiPrecision = ir.selectedPrecisions.length > 1; const officialByGroup = new Map(); for (const entry of entries) { if (!entry.visible) continue; - const groupKey = multiPrecision ? entry.key : entry.hw; + // One label per hardware and, under a power comparison, per + // sibling series: the measured line and its boundary / pool + // lines each say which one they are, instead of the longest + // line taking the hardware's only label. + const groupKey = multiPrecision ? entry.key : `${entry.hw}::${entry.variant}`; const previous = officialByGroup.get(groupKey); if (!previous || entry.points.length > previous.points.length) { officialByGroup.set(groupKey, entry); @@ -2582,39 +2934,58 @@ const ScatterGraph = React.memo( const officialSeries: LineLabelSeries[] = [ ...officialByGroup.values(), - ].map((entry) => ({ - key: entry.key, - seriesId: entry.hw, - label: lineLabelText( - entry.hw, - entry.precision, - multiPrecision, - modelLabel, - entry.points, - ), - color: ir.getCssColor(ir.resolveColor(entry.hw)), - points: entry.points, - keepVisibleOnCollision: entry.points.length === 1, - })); + ].map((entry) => { + const identity = lineLabelIdentity(entry.hw, entry.points); + const suffix = lineLabelSuffix(identity, entry.points); + lineLabelMeta.set(entry.key, { variantId: identity.variantId, suffix, runTag: '' }); + return { + key: entry.key, + seriesId: identity.seriesId, + label: `${lineLabelText( + entry.hw, + entry.precision, + multiPrecision, + modelLabel, + entry.points, + )}${suffix}`, + color: ir.getCssColor(ir.resolveColor(entry.hw)), + points: entry.points, + }; + }); + // Runs drawing the same hardware need a run tag on their pills. + const overlayRunsByHw = new Map>(); + for (const group of Object.values(displayedOverlayRooflines)) { + if (!ir.activeOverlayHwTypes.has(group.hwKey)) continue; + if (!overlayRunsByHw.has(group.hwKey)) overlayRunsByHw.set(group.hwKey, new Set()); + overlayRunsByHw.get(group.hwKey)!.add(group.runIndex); + } const overlaySeries: LineLabelSeries[] = Object.entries( displayedOverlayRooflines, ).flatMap(([overlayKey, group]) => { if (!ir.activeOverlayHwTypes.has(group.hwKey)) return []; const info = unofficialRunInfos[group.runIndex]; const precision = group.points[0]?.precision ?? ''; - const runLabel = info - ? getInferenceRunLabel(`✕ ${info.branch || `run ${info.id}`}`, group.points) - : ''; + const hardwareLabel = lineLabelText( + group.hwKey, + precision, + multiPrecision, + modelLabel, + group.points, + ); + const sharesHardware = (overlayRunsByHw.get(group.hwKey)?.size ?? 0) > 1; + const runTag = info && sharesHardware ? overlayRunTag(info) : ''; const label = info - ? multiPrecision - ? `${runLabel} ${getPrecisionLabel(precision as Precision)}` - : runLabel - : lineLabelText(group.hwKey, precision, multiPrecision, modelLabel, group.points); + ? getOverlayLineLabel(hardwareLabel, info, sharesHardware) + : hardwareLabel; + const identity = lineLabelIdentity(group.hwKey, group.points); + const suffix = lineLabelSuffix(identity, group.points); + const key = `overlay-${overlayKey}`; + lineLabelMeta.set(key, { variantId: identity.variantId, suffix, runTag }); return [ { - key: `overlay-${overlayKey}`, - seriesId: group.hwKey, - label, + key, + seriesId: identity.seriesId, + label: `${label}${suffix}`, color: overlayRunColor(group.runIndex), points: group.points, }, @@ -2637,16 +3008,19 @@ const ScatterGraph = React.memo( const labeledKeys = new Set(lineLabels.map((label) => label.key)); for (const entry of entries) { if (labeledKeys.has(entry.key)) continue; + const identity = lineLabelIdentity(entry.hw, entry.points); + const suffix = lineLabelSuffix(identity, entry.points); + lineLabelMeta.set(entry.key, { variantId: identity.variantId, suffix, runTag: '' }); lineLabels.push({ key: entry.key, - seriesId: entry.hw, - label: lineLabelText( + seriesId: identity.seriesId, + label: `${lineLabelText( entry.hw, entry.precision, multiPrecision, modelLabel, entry.points, - ), + )}${suffix}`, color: ir.getCssColor(ir.resolveColor(entry.hw)), x: xScale(entry.points[0].x), y: yScale(entry.points[0].y), @@ -2663,20 +3037,35 @@ const ScatterGraph = React.memo( } renderLineLabels(zoomGroup, lineLabels, { - seriesAttribute: 'data-hw-key', - iconFor: (label) => getLineLabelVendorIcon(label.seriesId), + seriesAttribute: 'data-series-id', + iconFor: (label) => getLineLabelVendorIcon(lineLabelHardwareKey(label.seriesId)), configureGroup: (labelGroup, label) => { labelGroup .attr('data-visible', label.visible ? '1' : '0') + // Legend hover and filter sync key labels by hardware alone; + // the variant names the comparison sibling ('' for the base). + .attr('data-hw-key', lineLabelHardwareKey(label.seriesId)) + .attr('data-power-variant', lineLabelMeta.get(label.key)?.variantId ?? '') .select('.ll-bg') .attr('opacity', 0.95); }, configureText: (text, label) => { - const config = getHardwareConfig(label.seriesId, modelLabel); + const config = getHardwareConfig(lineLabelHardwareKey(label.seriesId), modelLabel); + // Parse the hardware part without the variant suffix, which gets + // its own segment so the engine is still matched at the end. + const meta = lineLabelMeta.get(label.key); + const suffix = meta?.suffix ?? ''; + const runTag = meta?.runTag ?? ''; + let coreLabel = suffix ? label.label.slice(0, -suffix.length) : label.label; + if (runTag) coreLabel = coreLabel.slice(0, -runTag.length); + // Overlay pills lead with the run marker; the hardware behind it is + // parsed like an official pill so the GPU name stays bold. + const marker = coreLabel.startsWith(OVERLAY_LABEL_MARKER) ? OVERLAY_LABEL_MARKER : ''; + coreLabel = coreLabel.slice(marker.length); const hardwareLabel = getDisplayLabel(config); const isHardwareLabel = - label.label === hardwareLabel || label.label.startsWith(`${config.label} `); - const remainingLabel = isHardwareLabel ? label.label.slice(config.label.length) : ''; + coreLabel === hardwareLabel || coreLabel.startsWith(`${config.label} `); + const remainingLabel = isHardwareLabel ? coreLabel.slice(config.label.length) : ''; // Use this curve's resolved suffix, not the generic hwKey label: // official and overlay curves can share a key but differ by run. const engineLabel = @@ -2686,8 +3075,18 @@ const ScatterGraph = React.memo( engineLabel && remainingLabel.endsWith(engineLabel) ? remainingLabel.slice(0, -engineLabel.length) : remainingLabel; + const markerSegments = marker + ? [{ className: 'll-marker', text: marker, fill: 'white', weight: '600' }] + : []; + const runSegments = runTag + ? [{ className: 'll-run', text: runTag, fill: '#d1d5db', weight: '400' }] + : []; + const variantSegments = suffix + ? [{ className: 'll-variant', text: suffix, fill: 'white', weight: '500' }] + : []; const segments = isHardwareLabel ? [ + ...markerSegments, { className: 'll-gpu', text: config.label, fill: 'white', weight: '700' }, ...(precisionLabel ? [ @@ -2709,14 +3108,19 @@ const ScatterGraph = React.memo( }, ] : []), + ...runSegments, + ...variantSegments, ] : [ + ...markerSegments, { className: 'll-plain', - text: label.label, + text: coreLabel, fill: 'white', weight: '600', }, + ...runSegments, + ...variantSegments, ]; text .selectAll('tspan') @@ -2725,7 +3129,24 @@ const ScatterGraph = React.memo( .attr('class', (segment) => segment.className) .attr('fill', (segment) => segment.fill) .attr('font-weight', (segment) => segment.weight) + .attr('x', null) + .attr('dy', null) .text((segment) => segment.text); + // Keep the framework and role visible when a pill is wider than + // the mobile plot, without shrinking its text or dropping fields. + const textX = Number(text.attr('x') ?? 0); + const maxLineWidth = ctx.width - textX - 10; + let lineWidth = 0; + text.selectAll('tspan').each(function () { + const width = this.getComputedTextLength(); + if (lineWidth > 0 && lineWidth + width > maxLineWidth) { + d3.select(this) + .attr('x', textX) + .attr('dy', CHART_TYPE.lineLabel + 3); + lineWidth = 0; + } + lineWidth += width; + }); }, }); // Labels can be joined independently of the Pareto display pass. @@ -2826,11 +3247,11 @@ const ScatterGraph = React.memo( { key: string; seriesId: string; points: InferenceData[] } >(); for (const [key, points] of Object.entries(displayedRooflines)) { - const hardware = key.split('_').slice(0, -1).join('_'); - const precision = key.split('_').pop()!; + const { hw: hardware, precision, variant } = parseScatterSeriesKey(key); if ( !ir.effectiveActiveHwTypes.has(hardware) || - !ir.selectedPrecisions.includes(precision) + !ir.selectedPrecisions.includes(precision) || + ir.hiddenPowerVariants.has(variant ?? '') ) { continue; } @@ -2838,12 +3259,12 @@ const ScatterGraph = React.memo( const singleDate = pointsByDate.size === 1; for (const [date, datePoints] of pointsByDate) { const entryKey = singleDate ? key : `${key}__${encodeURIComponent(date)}`; - const groupKey = multiPrecision ? entryKey : hardware; + const groupKey = multiPrecision ? entryKey : `${hardware}::${variant ?? ''}`; const previous = bestByGroup.get(groupKey); if (!previous || datePoints.length > previous.points.length) { bestByGroup.set(groupKey, { key: entryKey, - seriesId: hardware, + seriesId: lineLabelIdentity(hardware, datePoints).seriesId, points: datePoints, }); } @@ -2854,7 +3275,6 @@ const ScatterGraph = React.memo( ...entry, label: '', color: '', - keepVisibleOnCollision: entry.points.length === 1, }), ); const overlaySeries: LineLabelSeries[] = Object.entries( @@ -2864,7 +3284,7 @@ const ScatterGraph = React.memo( ? [ { key: `overlay-${overlayKey}`, - seriesId: group.hwKey, + seriesId: lineLabelIdentity(group.hwKey, group.points).seriesId, label: '', color: '', points: group.points, @@ -2898,7 +3318,8 @@ const ScatterGraph = React.memo( interactionRef.current.getCssColor( interactionRef.current.resolveColor(d.hwKey as string), ), - getOpacity: (d) => (interactionRef.current.isPointVisible(d) ? 1 : 0), + getOpacity: (d) => + interactionRef.current.isPointVisible(d) ? pointOpacityForVariant(d) : 0, getPointerEvents: (d) => (interactionRef.current.isPointVisible(d) ? 'auto' : 'none'), hideLabels: !showPointLabels || showGradientLabels, // Concurrency (C=) is appended only when the advanced @@ -2908,6 +3329,7 @@ const ScatterGraph = React.memo( dataAttrs: { 'hw-key': (d) => String(d.hwKey), precision: (d) => d.precision, + 'power-variant': (d) => d.powerVariant?.id ?? '', // Lets the agentic coach mark pick an anchor out of the DOM // without knowing anything about React state. 'benchmark-type': (d) => d.benchmark_type ?? '', @@ -2986,6 +3408,7 @@ const ScatterGraph = React.memo( points: InferenceData[]; stroke: string; runIndex: number; + variant: string | null; } const ovEntries: OvEntry[] = []; Object.entries(displayedOverlayRooflines).forEach(([key, group]) => { @@ -2997,6 +3420,7 @@ const ScatterGraph = React.memo( // Color by run — same palette entry the legend uses, so they match. stroke: overlayRunColor(group.runIndex), runIndex: group.runIndex, + variant: group.variant, }); } }); @@ -3014,9 +3438,21 @@ const ScatterGraph = React.memo( .attr('fill', 'none') .attr('stroke', (d) => d.stroke) .attr('stroke-width', 2) - .attr('stroke-dasharray', (d) => overlayRooflineDasharray(d.runIndex)) + .attr('data-power-variant', (d) => d.variant) + // The run keeps its colour; a comparison sibling takes the + // variant dash so it reads like its official counterpart. + .attr('stroke-dasharray', (d) => + d.variant + ? powerVariantDashById(d.variant) + : overlayRooflineDasharray(d.runIndex), + ) .attr('d', (d) => lineGen(d.points)) - .style('filter', null); + .style('filter', null) + // Comparison rows hidden from the legend (the decoration effect + // keeps this in step with later toggles). + .style('opacity', (d) => + interactionRef.current.hiddenPowerVariants.has(d.variant ?? '') ? 0 : null, + ); // Overlay X-shape points — index-keyed so every point renders const overlayPoints = zoomGroup @@ -3052,7 +3488,7 @@ const ScatterGraph = React.memo( overlayPoints.each(function (d) { const visible = interactionRef.current.isOverlayPointVisible(d); d3.select(this) - .style('opacity', visible ? 1 : 0) + .style('opacity', visible ? pointOpacityForVariant(d) : 0) .style('pointer-events', visible ? 'auto' : 'none'); }); overlayPoints @@ -3136,6 +3572,9 @@ const ScatterGraph = React.memo( y: point.y, overlay: true, }); + // The shared helper has just rendered the pinned content into + // this element and pinned it via `handle`. + attachPowerTraceAction(ctx.tooltipElement, point, true); }, }); }, @@ -3406,10 +3845,12 @@ const ScatterGraph = React.memo( xLabel, yLabel, selectedYAxisMetric, + powerCompareBaseId, isMeasuredEnergyAxis, chartDefinition, locale, drawPerfRuler, + attachPowerTraceAction, ]); // Layers handle for the decoration effect — lets it re-run individual @@ -3488,7 +3929,9 @@ const ScatterGraph = React.memo( zoomGroup.selectAll('.dot-group').each(function (d) { const point = d3.select(this); const visible = ir.isPointVisible(d); - point.style('opacity', visible ? 1 : 0).style('pointer-events', visible ? 'auto' : 'none'); + point + .style('opacity', visible ? pointOpacityForVariant(d) : 0) + .style('pointer-events', visible ? 'auto' : 'none'); const color = (showGradientLabels && gradientColorByPoint.get(d)) || ir.getCssColor(ir.resolveColor(d.hwKey as string)); @@ -3507,9 +3950,20 @@ const ScatterGraph = React.memo( zoomGroup.selectAll('.unofficial-overlay-pt').each(function (d) { const visible = ir.isOverlayPointVisible(d); d3.select(this) - .style('opacity', visible ? 1 : 0) + .style('opacity', visible ? pointOpacityForVariant(d) : 0) .style('pointer-events', visible ? 'auto' : 'none'); }); + // Overlay rooflines are only drawn for active overlay hardware; a + // comparison row hidden from the legend is the one visibility toggle + // they answer to here. + zoomGroup.selectAll('.overlay-roofline-path').each(function () { + const roofline = d3.select(this); + if (ir.hiddenPowerVariants.has(this.dataset.powerVariant ?? '')) { + roofline.style('opacity', 0); + } else { + roofline.style('opacity', null); + } + }); // Rooflines: visibility and solid-stroke recolor as direct writes. Keep // gradient url references intact and never touch animated path geometry. @@ -3519,7 +3973,9 @@ const ScatterGraph = React.memo( if (!hw || !precision) return; const roofline = d3.select(this); const visible = - ir.effectiveActiveHwTypes.has(hw) && ir.selectedPrecisions.includes(precision); + ir.effectiveActiveHwTypes.has(hw) && + ir.selectedPrecisions.includes(precision) && + !ir.hiddenPowerVariants.has(this.dataset.powerVariant ?? ''); roofline.style('opacity', visible ? 1 : 0); const stroke = roofline.attr('stroke'); if (stroke && !stroke.startsWith('url(')) { @@ -3553,6 +4009,7 @@ const ScatterGraph = React.memo( (this as SVGGElement).dataset, ir.effectiveActiveHwTypes, ir.selectedPrecisions, + ir.activeOverlayHwTypes, ); }); }, [ @@ -3711,7 +4168,9 @@ const ScatterGraph = React.memo( // brings a hidden ruler back); curves whose paths left the DOM // entirely are truly gone from the data, so prune each ruler (and the // draft) that references one. `prunePerfRulers` bails out with the - // same reference when nothing changed. + // same reference when nothing changed. Share-link rulers still + // pending are not state yet, so prune cannot touch them; drawPerfRuler + // above committed those whose curves now exist. setPerfRulerState((prev) => prunePerfRulers(prev, (cls) => !display.zoomGroup.select(`.${CSS.escape(cls)}`).empty()), ); @@ -3973,6 +4432,9 @@ const ScatterGraph = React.memo( ) : null, })), + // Comparison series (`i_pcompare`): one dash-swatch row per + // boundary / role, toggling that series across every hardware. + ...powerVariantLegendItems, ]} disableActiveSort={false} isLegendExpanded={isLegendExpanded} @@ -4126,7 +4588,10 @@ const ScatterGraph = React.memo( // the pre-paint decoration effect then removes the rulers // and the curve hit strokes before the next frame (no // lingering lines after toggle-off). - if (!checked) setPerfRulerState(clearPerfRulers); + if (!checked) { + setPerfRulerState(clearPerfRulers); + persistedRulers?.discardPending(); + } track('latency_perf_ruler_toggled', { enabled: checked }); }, }, @@ -4174,6 +4639,7 @@ const ScatterGraph = React.memo( count: perfRulerState.rulers.length, }); setPerfRulerState(clearPerfRulers); + persistedRulers?.discardPending(); }, }, ] @@ -4226,6 +4692,15 @@ const ScatterGraph = React.memo( } /> )} + {powerTelemetryPoint === null ? null : ( + { + if (!open) setPowerTelemetryPoint(null); + }} + /> + )} {fixedLogPointId === null ? null : ( { - const ay = getNestedYValue(a, yPath); - const by = getNestedYValue(b, yPath); + const ay = a.powerVariant ? a.y : getNestedYValue(a, yPath); + const by = b.powerVariant ? b.y : getNestedYValue(b, yPath); return yAscending ? ay - by : by - ay; }); } diff --git a/packages/app/src/components/inference/ui/line-label-layer.test.ts b/packages/app/src/components/inference/ui/line-label-layer.test.ts index 33bb77256..e3eda7273 100644 --- a/packages/app/src/components/inference/ui/line-label-layer.test.ts +++ b/packages/app/src/components/inference/ui/line-label-layer.test.ts @@ -14,17 +14,12 @@ interface Point { y: number; } -const series = ( - key: string, - points: Point[], - keepVisibleOnCollision = false, -): LineLabelSeries => ({ +const series = (key: string, points: Point[]): LineLabelSeries => ({ key, seriesId: key, label: key, color: '#000', points, - keepVisibleOnCollision, }); const identity = (value: number) => value; diff --git a/packages/app/src/components/inference/ui/line-label-layer.ts b/packages/app/src/components/inference/ui/line-label-layer.ts index 6da0a7a23..a3b635dfa 100644 --- a/packages/app/src/components/inference/ui/line-label-layer.ts +++ b/packages/app/src/components/inference/ui/line-label-layer.ts @@ -16,7 +16,6 @@ export interface LineLabelSeries { label: string; color: string; points: readonly TPoint[]; - keepVisibleOnCollision?: boolean; } export interface LineLabelPlacement { @@ -26,6 +25,12 @@ export interface LineLabelPlacement { color: string; x: number; y: number; + /** + * `placeLineLabels` always emits `true`: every series it is given keeps its + * pill, overlapping if it must. The only producer of `false` is the caller's + * de-duplication pass, which keeps a hidden data-join entry for a curve that + * lost the one-label-per-hardware contest (GH #470). + */ visible: boolean; } @@ -146,16 +151,16 @@ interface PillLayoutItem { * its anchor, and both mirrors together. Every candidate is clamped into * `bounds` before the overlap test, so nothing leaves the plot. When every * mirrored candidate collides, nearby rows are tried before the default spot - * is kept: an overlapped label is still - * better than a missing one, and the fallback matches what the anchor pass - * already tolerates for pinned anchors. + * is kept: an overlapped label is still better than a missing one, and the + * fallback matches what the anchor pass already tolerates. * * With no bounds — a chart that clips nothing — the anchor offset is applied * unchanged and the collision pass is skipped, preserving that chart's * existing layout. * * Hidden pills get their default transform and occupy no space, so a label - * that later becomes visible reappears where the anchor pass put it. + * that later becomes visible reappears where the anchor pass put it. Only the + * caller's de-duplication pass hides pills; the anchor pass never does. */ function layoutPills( items: readonly PillLayoutItem[], @@ -319,6 +324,21 @@ function lineCandidates( return candidates; } +/** + * Anchor one pill per series along its line. + * + * Each series tries `ANCHOR_SLOTS` fractions along its own points, rotated by + * its index so converging curves spread out instead of stacking at the + * endpoint. A series that finds a clear slot takes it. A series that finds none + * is deferred and placed afterwards on its least crowded slot, so it never + * steals a clear slot from a series that could have used it. + * + * Every series gets a visible pill. The overlap that survives here is resolved + * by `layoutPills`, which runs later with the pills' real measured boxes; the + * crude nominal box used here is far too small to decide that a label is + * unplaceable — a rendered pill is routinely two to three times + * `collisionWidth`. + */ export function placeLineLabels( series: readonly LineLabelSeries[], xScale: (value: number) => number, @@ -346,6 +366,35 @@ export function placeLineLabels( Math.abs(other.y - y) < collisionHeight && Math.abs(other.x - x) < other.halfW + labelHalfWidth, ); + /** + * Nominal overlap area against the labels already placed. The same crude box + * model as `collides`, scored instead of thresholded, so a slot that clips one + * neighbour is preferred over one that sits on three. + */ + const collisionCost = (x: number, y: number) => + placed.reduce((cost, other) => { + const dx = other.halfW + labelHalfWidth - Math.abs(other.x - x); + const dy = collisionHeight - Math.abs(other.y - y); + return dx > 0 && dy > 0 ? cost + dx * dy : cost; + }, 0); + + const emit = (entry: LineLabelSeries, point: TPoint) => { + const x = xScale(point.x); + const y = yScale(point.y); + placed.push({ x, y, halfW: labelHalfWidth }); + result.push({ + key: entry.key, + seriesId: entry.seriesId, + label: entry.label, + color: entry.color, + x, + y, + visible: true, + }); + }; + + /** Series with no clear slot, deferred to a second pass — see below. */ + const crowded: { entry: LineLabelSeries; candidates: TPoint[] }[] = []; for (const [seriesIndex, entry] of sorted.entries()) { if (entry.points.length === 0) continue; @@ -376,35 +425,37 @@ export function placeLineLabels( const candidate = candidates.find((point) => !collides(xScale(point.x), yScale(point.y))); if (candidate) { - const x = xScale(candidate.x); - const y = yScale(candidate.y); - placed.push({ x, y, halfW: labelHalfWidth }); - result.push({ - key: entry.key, - seriesId: entry.seriesId, - label: entry.label, - color: entry.color, - x, - y, - visible: true, - }); + emit(entry, candidate); continue; } - const fallback = entry.points[0]; - const x = xScale(fallback.x); - const y = yScale(fallback.y); - const visible = entry.keepVisibleOnCollision === true; - if (visible) placed.push({ x, y, halfW: labelHalfWidth }); - result.push({ - key: entry.key, - seriesId: entry.seriesId, - label: entry.label, - color: entry.color, - x, - y, - visible, - }); + // No clear slot. Defer rather than claim one now: a series that is going to + // overlap something must not take a slot a later series could have had to + // itself. + crowded.push({ entry, candidates }); + } + + // Every series keeps a pill. `layoutPills` runs after this with the real + // measured boxes and can still mirror it, shift it a row and clamp it into the + // plot — "an overlapped label is still better than a missing one". Emitting the + // crowded ones last also hands that pass the clean labels first, so the crowded + // ones do the moving. + // + // This is where the chart stopped promising that line labels never overlap + // (#132 introduced the drop as the only way to honour that, #434 restated it). + // The promise was worth less than it cost: a dropped pill is silent, and with + // line labels on, PNG export omits the legend, so the series loses its only + // identifier. An overlapping pill at least announces itself. + for (const { entry, candidates } of crowded) { + emit( + entry, + candidates.reduce((best, point) => + collisionCost(xScale(point.x), yScale(point.y)) < + collisionCost(xScale(best.x), yScale(best.y)) + ? point + : best, + ), + ); } return result; diff --git a/packages/app/src/components/inference/ui/line-label-visibility.test.ts b/packages/app/src/components/inference/ui/line-label-visibility.test.ts index 80e17b049..3b4a075d8 100644 --- a/packages/app/src/components/inference/ui/line-label-visibility.test.ts +++ b/packages/app/src/components/inference/ui/line-label-visibility.test.ts @@ -53,6 +53,26 @@ describe('labelOpacityForActiveState', () => { }); }); +describe('labelOpacityForActiveState with ?unofficialrun= overlays', () => { + const official = new Set(['gb200_dynamo-sglang']); + const precisions = ['fp8']; + + it('hides an overlay label when its overlay hardware row is off, even if the official row is on', () => { + expect( + labelOpacityForActiveState( + { + hwKey: 'gb200_dynamo-sglang', + lineKey: 'overlay-gb200_dynamo-sglang_fp8_run1', + visible: '1', + }, + official, + precisions, + new Set(['gb300_dynamo-sglang']), + ), + ).toBe(0); + }); +}); + describe('labelOpacityForHover', () => { it('lights up the kept label for the hovered hardware', () => { expect(labelOpacityForHover({ hwKey: 'b300_sglang', visible: '1' }, 'b300_sglang')).toBe(1); diff --git a/packages/app/src/components/inference/ui/line-label-visibility.ts b/packages/app/src/components/inference/ui/line-label-visibility.ts index a8b34b810..9dfa8e878 100644 --- a/packages/app/src/components/inference/ui/line-label-visibility.ts +++ b/packages/app/src/components/inference/ui/line-label-visibility.ts @@ -25,6 +25,8 @@ export interface LabelAttrs { /** `data-hw-key` — base hardware key, shared across a hw's curves. */ hwKey?: string; + /** `data-line-key` — `overlay-…` marks an unofficial-run curve's label. */ + lineKey?: string; /** `data-precision` — set on parallelism labels, absent on line labels. */ precision?: string; /** `data-visible` — `'1'`/`'0'`; only line labels set this. */ @@ -50,16 +52,24 @@ export const labelOpacityForHover = (attrs: LabelAttrs, hoveredHwKey: string): 0 * filter-change sync effect. Line labels (no precision) show when their * hardware is active **and** the render kept them; parallelism labels show when * their hardware is active and their precision is selected. + * + * An `?unofficialrun=` overlay curve answers to the overlay legend rows, not + * the official ones: its label follows `activeOverlayHwTypes`, so soloing an + * official hardware no longer hides the overlay pills of every other hardware + * (and hiding an official row keeps its overlay twin labelled). */ export const labelOpacityForActiveState = ( attrs: LabelAttrs, activeHwTypes: ReadonlySet, selectedPrecisions: readonly string[], + activeOverlayHwTypes?: ReadonlySet, ): 0 | 1 => { const { hwKey, precision } = attrs; if (!hwKey) return 0; + const isOverlay = attrs.lineKey?.startsWith('overlay-') ?? false; + const active = isOverlay && activeOverlayHwTypes ? activeOverlayHwTypes : activeHwTypes; if (!precision) { - return activeHwTypes.has(hwKey) && renderKept(attrs) ? 1 : 0; + return active.has(hwKey) && renderKept(attrs) ? 1 : 0; } - return activeHwTypes.has(hwKey) && selectedPrecisions.includes(precision) ? 1 : 0; + return active.has(hwKey) && selectedPrecisions.includes(precision) ? 1 : 0; }; diff --git a/packages/app/src/components/inference/utils.ts b/packages/app/src/components/inference/utils.ts index 5284aac3b..1794cd585 100644 --- a/packages/app/src/components/inference/utils.ts +++ b/packages/app/src/components/inference/utils.ts @@ -8,8 +8,15 @@ import { getGpuSpecs, type TcoBasis } from '@/lib/constants'; import chartDefinitions from '@/components/inference/metric-registry'; import { resolveXAxisField } from '@/components/inference/utils/resolveXAxisField'; import { remapInferencePoint } from '@/lib/chart-utils'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; -import type { ChartDefinition, ClippedInferenceData, InferenceData, YAxisMetricKey } from './types'; +import type { + ChartDefinition, + ClippedInferenceData, + InferenceData, + PowerCompare, + YAxisMetricKey, +} from './types'; import type { XAxisMode } from './hooks/useChartData'; /** @@ -145,6 +152,7 @@ export function processOverlayChartData( selectedXAxisMode?: XAxisMode; restrictToNormalizedFrontier?: boolean; tcoBasis?: TcoBasis; + powerCompare?: PowerCompare; }, ): InferenceData[] { return processOverlayChartDataWithClipping( @@ -171,6 +179,8 @@ export function processOverlayChartDataWithClipping( selectedXAxisMode?: XAxisMode; restrictToNormalizedFrontier?: boolean; tcoBasis?: TcoBasis; + /** Sibling boundary / role series, mirroring the official path in useChartData. */ + powerCompare?: PowerCompare; }, ): ProcessedChartData { const chartDef = (chartDefinitions as ChartDefinition[]).find((d) => d.chartType === chartType); @@ -222,9 +232,13 @@ export function processOverlayChartDataWithClipping( // for the natural axis and for agentic (long TTFTs are normal there). const isTtftX = xAxisField.endsWith('_ttft'); - const processedData = sourceData - .filter((d) => metricKey in d) - .map((d) => remapInferencePoint(d, metricKey, xAxisField)); + const processedData = expandPowerCompareSeries( + sourceData + .filter((d) => metricKey in d) + .map((d) => remapInferencePoint(d, metricKey, xAxisField)), + selectedYAxisMetric, + options?.powerCompare ?? 'none', + ); // The normalized metric is derived from persisted request traces, which an // unofficial overlay does not have. An all-false canonical stamp prevents a diff --git a/packages/app/src/components/inference/utils/best-series-per-sku.ts b/packages/app/src/components/inference/utils/best-series-per-sku.ts index 6799b9631..2a35f0b9c 100644 --- a/packages/app/src/components/inference/utils/best-series-per-sku.ts +++ b/packages/app/src/components/inference/utils/best-series-per-sku.ts @@ -40,7 +40,9 @@ export function bestSeriesPerSku(points: InferenceData[], direction: Direction): const bySku = new Map>(); const featured = new Set(); for (const point of points) { - if (!isFrontierEligible(point) || !Number.isFinite(point.y)) continue; + // Comparison clones re-plot the same configs at another boundary or role; + // the best series per SKU is judged on the selected metric alone. + if (point.powerVariant || !isFrontierEligible(point) || !Number.isFinite(point.y)) continue; const sku = baseSku(point); const key = String(point.hwKey); if (point.framework === 'tilert') featured.add(key); diff --git a/packages/app/src/components/inference/utils/point-identity.ts b/packages/app/src/components/inference/utils/point-identity.ts index 2c975593e..0cc8bbed8 100644 --- a/packages/app/src/components/inference/utils/point-identity.ts +++ b/packages/app/src/components/inference/utils/point-identity.ts @@ -23,9 +23,47 @@ export function scatterPointConfigId(point: InferenceData): string { // Agentic series omit spec decoding from hwKey so one curve can mix methods. // It remains point identity to avoid collapsing overlapping MTP/STP results. key += agenticSpecDecodingKeySuffix(point); + // Comparison clones share every config field with their base point. + if (point.powerVariant) key += `|variant-${point.powerVariant.id}`; return key; } +/** + * Comparison-series suffix inside a scatter series key. Letters, digits and + * dashes only, so the key stays a valid CSS class token (the perf ruler and + * `i_rulers` address rooflines by class) and needs no escaping. + */ +const SERIES_VARIANT_DELIMITER = '-v-'; + +/** + * Identity of one drawn series: hardware key, precision and, on a power + * comparison, the boundary or role variant. Rooflines, frontiers, line labels + * and the perf ruler all key on this string. + */ +export function scatterSeriesKey( + point: Pick, +): string { + const base = `${point.hwKey}_${point.precision}`; + return point.powerVariant ? `${base}${SERIES_VARIANT_DELIMITER}${point.powerVariant.id}` : base; +} + +export interface ScatterSeriesIdentity { + hw: string; + precision: string; + /** Comparison variant id (`gpu-provisioned`, `prefill`, …) or null for the base series. */ + variant: string | null; +} + +/** Inverse of `scatterSeriesKey`; hardware keys may themselves contain underscores. */ +export function parseScatterSeriesKey(key: string): ScatterSeriesIdentity { + const delimiter = key.indexOf(SERIES_VARIANT_DELIMITER); + const core = delimiter === -1 ? key : key.slice(0, delimiter); + const variant = delimiter === -1 ? null : key.slice(delimiter + SERIES_VARIANT_DELIMITER.length); + const parts = core.split('_'); + const precision = parts.pop() ?? ''; + return { hw: parts.join('_'), precision, variant }; +} + /** * Stable D3 join key for an official scatter point. * diff --git a/packages/app/src/components/inference/utils/power-compare.ts b/packages/app/src/components/inference/utils/power-compare.ts new file mode 100644 index 000000000..d0dea0b73 --- /dev/null +++ b/packages/app/src/components/inference/utils/power-compare.ts @@ -0,0 +1,326 @@ +/** + * Comparison series for the gated measured-power charts (`i_pcompare`). + * + * The selected metric stays the chart's base series. A comparison adds sibling + * series drawn from the SAME points — one per other power boundary + * (`boundaries`, PowerX Figures 2/3) or per worker role (`roles`, Figures 6/7) + * — so ScatterGraph can draw them with the hardware's colour and a per-variant + * dash through its ordinary series pipeline (frontier, Optimal Only, tooltip, + * table, CSV, `?unofficialrun=` overlay). A variant point is a clone of its + * base point with `y` remapped and `powerVariant` set; base points are left + * untouched so a chart without comparison is byte-identical to before. + * + * Variants exist only where the metric names a whole-deployment average + * (W per chip) or J per output token, because those are the only quantities + * every boundary and role publishes on a common axis; everywhere else the + * comparison yields no series and the control says so. + */ +import type { Locale } from '@/lib/i18n'; +import { + POWER_BASES, + POWER_BASIS_FIELDS, + POWER_BASIS_LABELS, + type PowerBasis, +} from '@/lib/power-basis'; + +import { getMeasuredMetricConfig } from '../measured-metric-config'; +import type { InferenceData, PowerCompare, PowerRole, PowerVariant } from '../types'; + +export const POWER_COMPARE_MODES = [ + 'none', + 'boundaries', + 'roles', +] as const satisfies readonly PowerCompare[]; + +export function parsePowerCompare(value: string | null | undefined): PowerCompare { + return value === 'boundaries' || value === 'roles' ? value : 'none'; +} + +/** Point fields a comparison series can plot; each is a `{ y, roof }` pair. */ +export type PowerSeriesField = keyof Pick< + InferenceData, + | 'measuredAvgPower' + | 'measuredPrefillAvgPower' + | 'measuredDecodeAvgPower' + | 'measuredJPerOutputToken' + | 'measuredDecodeJPerOutputToken' + | 'reconstructedPrefillJPerOutputToken' + | 'gpuProvisionedWatts' + | 'gpuProvisionedJPerOutputToken' + | 'utilityProvisionedWatts' + | 'utilityProvisionedJPerOutputToken' + | 'utilityModeledWatts' + | 'utilityModeledJPerOutputToken' +>; + +export interface PowerCompareSeries { + variant: PowerVariant; + field: PowerSeriesField; +} + +const ROLES = ['all', 'prefill', 'decode'] as const satisfies readonly PowerRole[]; + +const ROLE_WATT_FIELDS: Record = { + all: 'measuredAvgPower', + prefill: 'measuredPrefillAvgPower', + decode: 'measuredDecodeAvgPower', +}; +// Energy per output token per role. The prefill pool's own figure is per input +// token; `reconstructedPrefillJPerOutputToken` carries it onto the output-token +// axis (utils/role-energy.ts). +const ROLE_ENERGY_FIELDS: Record = { + all: 'measuredJPerOutputToken', + prefill: 'reconstructedPrefillJPerOutputToken', + decode: 'measuredDecodeJPerOutputToken', +}; + +const ROLE_LABELS: Record = { + all: { en: 'All GPUs', zh: '全部 GPU' }, + prefill: { en: 'Prefill GPUs', zh: '预填充 GPU' }, + decode: { en: 'Decode GPUs', zh: '解码 GPU' }, +}; + +/** SVG dash per variant; the base series and `all` stay solid. */ +const VARIANT_DASH: Record = { + 'gpu-measured': '', + 'gpu-provisioned': '8 4', + 'utility-provisioned': '3 3', + 'utility-modeled': '10 3 2 3', + all: '', + prefill: '7 3', + decode: '2 3', +}; + +/** + * The quantity a metric key plots on the comparison's common axis, or null + * when the key is not a whole-deployment average W/chip or J per output token. + */ +function comparableQuantity(metric: string): 'watts' | 'energy' | null { + const config = getMeasuredMetricConfig(metric); + if (!config) return null; + if (config.family === 'power') { + return config.scope === 'all' && config.statistic === 'average' && config.display === 'watts' + ? 'watts' + : null; + } + return config.scope === 'all' && config.denominator === 'output' && config.unit === 'joules' + ? 'energy' + : null; +} + +/** + * The base series' own identity under a comparison — the boundary or role the + * selected metric already plots — or null when the metric admits no comparison + * of that kind. + */ +export function powerCompareBase(metric: string, mode: PowerCompare): PowerVariant | null { + const config = getMeasuredMetricConfig(metric); + if (!config || mode === 'none') return null; + if (mode === 'boundaries') { + return comparableQuantity(metric) ? { kind: 'basis', id: config.basis } : null; + } + if (config.basis !== 'gpu-measured') return null; + if (config.family === 'power') { + return config.statistic === 'average' && config.display === 'watts' + ? { kind: 'role', id: config.scope } + : null; + } + // Role energy compares J per output token; a prefill J per input token axis + // has no decode counterpart. + return config.denominator === 'output' && config.unit === 'joules' && config.scope !== 'prefill' + ? { kind: 'role', id: config.scope } + : null; +} + +/** The sibling series a comparison adds to the selected metric (never the base itself). */ +export function powerCompareVariants(metric: string, mode: PowerCompare): PowerCompareSeries[] { + const base = powerCompareBase(metric, mode); + const config = getMeasuredMetricConfig(metric); + if (!base || !config) return []; + if (base.kind === 'basis') { + const quantity = comparableQuantity(metric); + if (!quantity) return []; + return POWER_BASES.filter((basis) => basis !== base.id).map((basis) => ({ + variant: { kind: 'basis', id: basis }, + field: + basis === 'gpu-measured' + ? quantity === 'watts' + ? 'measuredAvgPower' + : 'measuredJPerOutputToken' + : POWER_BASIS_FIELDS[basis][quantity], + })); + } + const fields = config.family === 'power' ? ROLE_WATT_FIELDS : ROLE_ENERGY_FIELDS; + return ROLES.filter((role) => role !== base.id).map((role) => ({ + variant: { kind: 'role', id: role }, + field: fields[role], + })); +} + +/** Whether choosing `mode` on `metric` draws anything. `none` is always available. */ +export function powerCompareAvailable(metric: string, mode: PowerCompare): boolean { + return mode === 'none' || powerCompareVariants(metric, mode).length > 0; +} + +/** + * Appends one clone per comparison series to `points` (already remapped onto + * the selected metric). A point lacking a variant's field contributes nothing + * to that series — never a 0 — so the availability rules of each boundary and + * role carry through unchanged. + */ +export function expandPowerCompareSeries( + points: readonly InferenceData[], + metric: string, + mode: PowerCompare, +): InferenceData[] { + const series = powerCompareVariants(metric, mode); + if (series.length === 0) return [...points]; + const result: InferenceData[] = [...points]; + for (const { variant, field } of series) { + for (const point of points) { + const value = point[field]; + if (!value || !Number.isFinite(value.y)) continue; + result.push({ ...point, y: value.y, roof: value.roof, powerVariant: variant }); + } + } + return result; +} + +/** Comparison mode implied by the variants present in a rendered point set. */ +export function inferPowerCompare(points: readonly InferenceData[]): PowerCompare { + for (const point of points) { + if (point.powerVariant?.kind === 'basis') return 'boundaries'; + if (point.powerVariant?.kind === 'role') return 'roles'; + } + return 'none'; +} + +/** Distinct variants in draw order: the base first, then siblings in canonical order. */ +export function powerVariantsInData( + points: readonly InferenceData[], + metric: string, +): PowerVariant[] { + const mode = inferPowerCompare(points); + if (mode === 'none') return []; + const present = new Set(points.map((point) => point.powerVariant?.id).filter(Boolean)); + const base = powerCompareBase(metric, mode); + const ordered: PowerVariant[] = + mode === 'boundaries' + ? POWER_BASES.map((id) => ({ kind: 'basis', id })) + : ROLES.map((id) => ({ kind: 'role', id })); + return ordered.filter((variant) => variant.id === base?.id || present.has(variant.id)); +} + +export function powerVariantId(variant: PowerVariant | null | undefined): string { + return variant?.id ?? ''; +} + +export function powerVariantDash(variant: PowerVariant | null | undefined): string { + return variant ? VARIANT_DASH[variant.id] : ''; +} + +export function powerVariantLabel(variant: PowerVariant, locale: Locale): string { + return variant.kind === 'basis' + ? POWER_BASIS_LABELS[variant.id][locale] + : ROLE_LABELS[variant.id][locale]; +} + +/** Short boundary names for in-chart line labels; the legend keeps the full names. */ +const BASIS_SHORT_LABELS: Record = { + 'gpu-measured': { en: 'Measured', zh: '实测' }, + 'gpu-provisioned': { en: 'TDP', zh: 'TDP' }, + 'utility-provisioned': { en: 'All-in', zh: '全站' }, + 'utility-modeled': { en: 'PUE modeled', zh: 'PUE 建模' }, +}; + +/** Suffix text a line label carries for one comparison series. */ +export function powerVariantShortLabel(variant: PowerVariant, locale: Locale): string { + return variant.kind === 'basis' + ? BASIS_SHORT_LABELS[variant.id][locale] + : ROLE_LABELS[variant.id][locale]; +} + +/** `700 W`, `1.37 kW`, `19.2 kW`: kilowatts from 1000 W, at most two decimals, zeros trimmed. */ +export function formatWatts(watts: number): string { + if (Math.abs(watts) >= 1000) { + return `${(watts / 1000).toFixed(2).replace(/\.?0+$/u, '')} kW`; + } + return `${Math.round(watts)} W`; +} + +/** + * The value a series holds at every point, or null when it varies. A + * provisioned boundary (TDP, all-in) is one number per hardware, so its line + * label can state it instead of sending the reader to the axis. Non-finite + * values are ignored; an empty series is null. + */ +export function flatSeriesValue(values: readonly number[], relTolerance = 0.005): number | null { + const finite = values.filter((value) => Number.isFinite(value)); + if (finite.length === 0) return null; + const first = finite[0]; + const tolerance = Math.abs(first) * relTolerance; + return finite.every((value) => Math.abs(value - first) <= tolerance) ? first : null; +} + +export interface PowerLineLabelOptions { + /** The selected metric's own series keeps the plain hardware label. */ + isBase: boolean; + locale: Locale; + /** Shared watts of a flat series (`flatSeriesValue`), appended after the name. */ + flatWatts?: number | null; +} + +const LINE_LABEL_SUFFIX_SEPARATOR = ' · '; + +/** Suffix appended to a comparison sibling's line label; '' for the base series. */ +export function powerLineLabelSuffix( + variant: PowerVariant | null | undefined, + opts: PowerLineLabelOptions, +): string { + if (opts.isBase || !variant) return ''; + const watts = + typeof opts.flatWatts === 'number' && Number.isFinite(opts.flatWatts) + ? ` ${formatWatts(opts.flatWatts)}` + : ''; + return `${LINE_LABEL_SUFFIX_SEPARATOR}${powerVariantShortLabel(variant, opts.locale)}${watts}`; +} + +const LINE_LABEL_SERIES_DELIMITER = '::'; + +/** + * Line-label series id: the hardware key for the base series (existing pinned + * anchors and hover hooks key on it) and `::` for a sibling. + */ +export function lineLabelSeriesId( + hw: string, + variant: PowerVariant | null | undefined, + isBase: boolean, +): string { + return isBase || !variant ? hw : `${hw}${LINE_LABEL_SERIES_DELIMITER}${variant.id}`; +} + +/** Inverse of `lineLabelSeriesId`: the hardware key behind a line-label series id. */ +export function lineLabelHardwareKey(seriesId: string): string { + const index = seriesId.indexOf(LINE_LABEL_SERIES_DELIMITER); + return index === -1 ? seriesId : seriesId.slice(0, index); +} + +/** Whether `metric` plots watts per chip, so a flat boundary's label can state its value. */ +export function metricPlotsWatts(metric: string): boolean { + const config = getMeasuredMetricConfig(metric); + return config?.family === 'power' && config.display === 'watts'; +} + +/** + * Series label for a table or CSV row: the point's variant, or the base + * series' identity when the chart is comparing and this is a base point. + */ +export function powerSeriesLabel( + point: Pick, + metric: string, + mode: PowerCompare, + locale: Locale, +): string { + const variant = point.powerVariant ?? powerCompareBase(metric, mode); + return variant ? powerVariantLabel(variant, locale) : ''; +} diff --git a/packages/app/src/components/inference/utils/powerCurves.test.ts b/packages/app/src/components/inference/utils/powerCurves.test.ts index a71f8bde2..dc0125649 100644 --- a/packages/app/src/components/inference/utils/powerCurves.test.ts +++ b/packages/app/src/components/inference/utils/powerCurves.test.ts @@ -94,8 +94,12 @@ describe('upper power envelope', () => { it('mirrors the boundary for latency and resolves tied coordinates deterministically', () => { const fast = point(1, 1, 350); const middle = point(8, 2, 700); + const plateau = point(16, 3, 700); const slow = point(32, 4, 950); - const samples = [slow, point(16, 3, 700), point(4, 2, 500), middle, { ...middle }, fast]; + // Measured watts: a tie at the running maximum is a repeat marker, so the + // plateau leaves the boundary and Optimal Only can collapse it; a repeated + // X keeps only its first vertex. + const samples = [slow, plateau, point(4, 2, 500), middle, { ...middle }, fast]; expect(upperPowerEnvelope(samples, false)).toEqual([fast, middle, slow]); expect( upperPowerEnvelope( @@ -103,6 +107,8 @@ describe('upper power envelope', () => { true, ).map((p) => p.y), ).toEqual([950, 700, 350]); + // A gauge keeps the plateau: it is part of the outer edge it draws. + expect(upperPowerEnvelope(samples, false, true)).toEqual([fast, middle, plateau, slow]); }); it('uses only finite positive coordinates and preserves singleton boundaries', () => { diff --git a/packages/app/src/components/inference/utils/powerCurves.ts b/packages/app/src/components/inference/utils/powerCurves.ts index 35146640a..13ac7209e 100644 --- a/packages/app/src/components/inference/utils/powerCurves.ts +++ b/packages/app/src/components/inference/utils/powerCurves.ts @@ -1,3 +1,4 @@ +import { isPowerBasisConfigKey } from '@/components/inference/metric-registry'; import type { InferenceData } from '@/components/inference/types'; import { isFrontierEligible, @@ -15,14 +16,40 @@ const POWER_CURVE_METRICS: ReadonlySet = new Set([ 'y_measuredDecodeAvgPower', 'y_measuredPowerPercentTdp', 'y_modeledChassisPowerPerGpu', + // Provisioned / modelled boundary gauges: same upper-envelope curve as measured watts. + 'y_gpuProvisionedWatts', + 'y_utilityProvisionedWatts', + 'y_utilityModeledWatts', ]); export function isPowerCurveMetric(metric: string): boolean { return POWER_CURVE_METRICS.has(metric); } +/** + * Power gauges whose curve is always the upper envelope and whose Optimal Only + * toggle only hides off-envelope markers: measured watts plus the provisioned / + * modelled boundary gauges that sit beside them. A Pareto corner would collapse + * a flat TDP series to one marker. The modelled chassis axis keeps its legacy + * Pareto behaviour. + */ export function isMeasuredPowerCurveMetric(metric: string): boolean { - return isPowerCurveMetric(metric) && metric !== 'y_modeledChassisPowerPerGpu'; + return ( + isPowerCurveMetric(metric) && + (metric !== 'y_modeledChassisPowerPerGpu' || isPowerBasisConfigKey(metric)) + ); +} + +/** + * Whether a drawn series is a provisioned or modelled gauge rather than + * telemetry: a boundary axis, or a boundary comparison clone of one. Gauges + * keep envelope ties (see `upperPowerEnvelope`); measured series, the measured + * boundary clone and role clones stay on the strict envelope. + */ +export function isPowerGaugeSeries(metric: string, sample: InferenceData | undefined): boolean { + const variant = sample?.powerVariant; + if (variant) return variant.kind === 'basis' && variant.id !== 'gpu-measured'; + return isPowerBasisConfigKey(metric); } /** No declared direction means there is no Pareto frontier to draw or filter by. */ @@ -45,14 +72,23 @@ export function chartFrontier( export function upperPowerEnvelope( points: readonly InferenceData[], maximizeX: boolean, + keepTies = false, ): InferenceData[] { const sorted = points .filter((point) => isFrontierEligible(point) && Number.isFinite(point.y) && point.y > 0) .sort((a, b) => (maximizeX ? b.x - a.x : a.x - b.x) || b.y - a.y); + // Measured telemetry keeps the strict envelope: a tie at the running maximum + // is a repeat marker that Optimal Only collapses. A provisioned or modelled + // gauge (`keepTies`) is flat by construction, so its ties stay on the + // boundary and the series draws across its tested x-range instead of one + // marker. Repeated X keeps only its first vertex so the smoothing never + // backtracks. let maxY = -Infinity; + let lastX = Number.NaN; const envelope = sorted.filter((point) => { - if (point.y <= maxY) return false; + if (point.y < maxY || (point.y === maxY && (!keepTies || point.x === lastX))) return false; maxY = point.y; + lastX = point.x; return true; }); return maximizeX ? envelope.toReversed() : envelope; diff --git a/packages/app/src/components/inference/utils/powerTimeline.test.ts b/packages/app/src/components/inference/utils/powerTimeline.test.ts new file mode 100644 index 000000000..409a74ad1 --- /dev/null +++ b/packages/app/src/components/inference/utils/powerTimeline.test.ts @@ -0,0 +1,21 @@ +import { describe, expect, it } from 'vitest'; + +import { prioritizeRuns } from './powerTimeline'; + +describe('prioritizeRuns', () => { + const requests = ['1', '2', '3', '4', '5'].map((runId) => ({ + runId, + prefix: '', + sources: [], + })); + + it('moves overlay runs ahead of official runs and keeps both orders', () => { + expect(prioritizeRuns(requests, new Set(['5', '3'])).map((request) => request.runId)).toEqual([ + '3', + '5', + '1', + '2', + '4', + ]); + }); +}); diff --git a/packages/app/src/components/inference/utils/powerTimeline.ts b/packages/app/src/components/inference/utils/powerTimeline.ts new file mode 100644 index 000000000..cff71ac26 --- /dev/null +++ b/packages/app/src/components/inference/utils/powerTimeline.ts @@ -0,0 +1,344 @@ +/** + * Joins chart points to the per-second GPU telemetry behind their measured + * average power (the PowerX "Timeline" display). + * + * Every validated row records its window in `power_audit`, whose `source` is + * `power_validation_.json`. Two collectors publish the telemetry: + * - single-node runners upload one `gpu_metrics_` CSV artifact per + * config, so the source names the artifact exactly; + * - Slurm / Dynamo runners upload one `power_audit_` bundle per + * sweep whose `LOGS/power/samples.csv` covers every concurrency; the API cuts + * it into one series per `power_validation_*.json` it contains and labels + * each with that `source`, so the same file name joins it to the row. + * Nothing is matched by hardware or concurrency; points whose telemetry is + * missing (expired artifact, another collector) are reported, not guessed. + */ +import type { + GpuPowerRole, + GpuPowerSeries, + GpuPowerSeriesResponse, +} from '@/components/gpu-power/power-series'; +import type { InferenceData } from '@/components/inference/types'; + +export const POWER_TIMELINE_METRIC_KEY = 'y_measuredPowerTimeline'; +const ARTIFACT_PREFIX = 'gpu_metrics_'; +const SOURCE_PATTERN = /^(?:.*\/)?power_validation_(?.+)\.json$/u; + +interface AuditedPoint { + power_audit?: { source?: string } | null; +} + +/** The `` of a point's `power_validation_.json` audit source. */ +export function telemetryNameForPoint(point: AuditedPoint): string | null { + const source = point.power_audit?.source; + if (!source) return null; + return SOURCE_PATTERN.exec(source)?.groups?.name ?? null; +} + +/** `gpu_metrics_` for a point, from its power-audit source. */ +export function telemetryArtifactForPoint(point: AuditedPoint): string | null { + const name = telemetryNameForPoint(point); + return name ? `${ARTIFACT_PREFIX}${name}` : null; +} + +/** + * Stable identity of a point's trace: run id plus audit name. Unique within a + * chart because the audit name carries the config and the concurrency. + */ +export function traceKeyForPoint(point: AuditedPoint & { run_url?: string }): string | null { + const runId = runIdFromUrl(point.run_url); + const name = telemetryNameForPoint(point); + return runId && name ? `${runId}:${name}` : null; +} + +/** Workflow run id from a GitHub Actions run URL. */ +export function runIdFromUrl(url: string | null | undefined): string | null { + return url?.match(/\/runs\/(?\d+)/u)?.groups?.runId ?? null; +} + +export function longestCommonPrefix(values: readonly string[]): string { + if (values.length === 0) return ''; + let prefix = values[0]; + for (const value of values) { + let end = 0; + while (end < prefix.length && end < value.length && prefix[end] === value[end]) end++; + prefix = prefix.slice(0, end); + if (prefix === '') break; + } + return prefix; +} + +export interface PowerTimelineRequest { + runId: string; + /** RESULT_FILENAME prefix shared by every wanted artifact of the run. */ + prefix: string; + /** Sorted, unique validation basenames expected by the displayed points. */ + sources: string[]; +} + +/** + * One request per workflow run, narrowed to the common RESULT_FILENAME prefix + * of the points' artifacts so a sweep of other models is not downloaded. + * Points without a run URL or an audit source plan nothing. + */ +export function planPowerTimelineRequests( + points: readonly InferenceData[], +): PowerTimelineRequest[] { + const byRun = new Map>(); + for (const point of points) { + const runId = runIdFromUrl(point.run_url); + const artifact = telemetryArtifactForPoint(point); + if (!runId || !artifact) continue; + if (!byRun.has(runId)) byRun.set(runId, new Set()); + byRun.get(runId)!.add(artifact); + } + return [...byRun.entries()] + .toSorted(([a], [b]) => a.localeCompare(b)) + .map(([runId, artifacts]) => { + const names = [...artifacts].toSorted(); + return { + runId, + prefix: longestCommonPrefix(names.map((name) => name.slice(ARTIFACT_PREFIX.length))), + sources: names.map((name) => `power_validation_${name.slice(ARTIFACT_PREFIX.length)}.json`), + }; + }); +} + +/** Run id half of a trace key (`traceKeyForPoint`). */ +export function traceKeyRunId(key: string | null | undefined): string | null { + const runId = key?.split(':')[0]; + return runId || null; +} + +/** + * Moves the request for `runId` to the front so it survives the per-chart run + * cap; the order of the other requests is kept. Returns the same array when + * nothing needs moving. + */ +export function prioritizeRun( + requests: PowerTimelineRequest[], + runId: string | null, +): PowerTimelineRequest[] { + const index = runId ? requests.findIndex((request) => request.runId === runId) : -1; + if (index <= 0) return requests; + return [requests[index], ...requests.slice(0, index), ...requests.slice(index + 1)]; +} + +/** + * Moves every request whose run is in `runIds` ahead of the others, keeping + * the relative order inside both groups. Unofficial-run overlays are loaded + * on purpose, so their telemetry must survive the per-chart run cap before + * official rows compete for the remaining slots. Returns the same array when + * nothing needs moving. + */ +export function prioritizeRuns( + requests: PowerTimelineRequest[], + runIds: ReadonlySet, +): PowerTimelineRequest[] { + if (runIds.size === 0) return requests; + const first = requests.filter((request) => runIds.has(request.runId)); + if (first.length === 0 || first.length === requests.length) return requests; + const rest = requests.filter((request) => !runIds.has(request.runId)); + const moved = first.some((request, index) => requests[index] !== request); + return moved ? [...first, ...rest] : requests; +} + +export interface PowerTimelineTrace { + /** `traceKeyForPoint(point)`. */ + key: string; + point: InferenceData; + runId: string; + series: GpuPowerSeries; + /** Validated measurement window (UTC ms) from `power_audit`, when recorded. */ + windowStartMs: number | null; + windowEndMs: number | null; +} + +/** + * Why a validated point has no trace: + * - `no-source`: the row predates `power_audit.source` (older schema), so no + * artifact can be named; + * - `no-run`: no workflow run URL to look in; + * - `run-not-fetched`: its run is not among the loaded responses (over the + * per-chart run limit, still loading, or the request failed); + * - `not-in-run`: the run was loaded but holds neither a matching + * `gpu_metrics_*` artifact nor a power-audit bundle with the point's + * validation file (expired, over the download cap, or another collector). + */ +export type MissingTraceReason = 'no-source' | 'no-run' | 'run-not-fetched' | 'not-in-run'; + +export interface MissingTrace { + point: InferenceData; + reason: MissingTraceReason; +} + +export interface PowerTimelineJoin { + traces: PowerTimelineTrace[]; + /** Points with a validated average but no telemetry trace, with the reason. */ + missing: MissingTrace[]; +} + +/** Attaches fetched series to points; order follows `points`. */ +export function joinPowerTimeline( + points: readonly InferenceData[], + responses: ReadonlyMap, +): PowerTimelineJoin { + const traces: PowerTimelineTrace[] = []; + const missing: MissingTrace[] = []; + for (const point of points) { + const runId = runIdFromUrl(point.run_url); + const name = telemetryNameForPoint(point); + if (!name) { + missing.push({ point, reason: 'no-source' }); + continue; + } + if (!runId) { + missing.push({ point, reason: 'no-run' }); + continue; + } + const response = responses.get(runId); + if (!response) { + missing.push({ point, reason: 'run-not-fetched' }); + continue; + } + // A bundle-cut series names the point's validation file; a per-config + // CSV artifact names the config itself. + const source = `power_validation_${name}.json`; + const artifact = `${ARTIFACT_PREFIX}${name}`; + const series = + response.series.find((entry) => entry.source === source) ?? + response.series.find((entry) => entry.source === undefined && entry.artifact === artifact); + if (!series) { + missing.push({ point, reason: 'not-in-run' }); + continue; + } + const audit = point.power_audit; + traces.push({ + key: `${runId}:${name}`, + point, + runId, + series, + windowStartMs: unixSecondsToMs(audit?.window_start_unix), + windowEndMs: unixSecondsToMs(audit?.window_end_unix), + }); + } + return { traces, missing }; +} + +function unixSecondsToMs(seconds: number | undefined): number | null { + return typeof seconds === 'number' && Number.isFinite(seconds) ? seconds * 1000 : null; +} + +/** Sample position relative to the validated window. */ +export type WindowPhase = 'before' | 'window' | 'after' | 'unknown'; + +export function windowPhase(trace: PowerTimelineTrace, timeMs: number): WindowPhase { + if (trace.windowStartMs === null || trace.windowEndMs === null) return 'unknown'; + if (timeMs < trace.windowStartMs) return 'before'; + if (timeMs > trace.windowEndMs) return 'after'; + return 'window'; +} + +/** Short config label: `TP8 · c64`, plus `PD` for disaggregated points. */ +export function traceConfigLabel(point: InferenceData): string { + const parts: string[] = []; + if (point.disagg) parts.push('PD'); + if (typeof point.tp === 'number' && point.tp > 0) parts.push(`TP${point.tp}`); + parts.push(`c${point.conc}`); + return parts.join(' · '); +} + +// ── Worker-role pools ──────────────────────────────────────────────────────── + +export type PowerPoolRole = GpuPowerRole | 'all'; + +export interface PowerPool { + role: PowerPoolRole; + /** Row indices into `series.power`, in series order. */ + rows: number[]; +} + +const POOL_ORDER: readonly PowerPoolRole[] = ['all', 'prefill', 'decode']; + +/** + * The GPU pools of a series by worker role, in `prefill`, `decode` order — + * only the roles that have at least one device. A series whose collector + * assigns no roles yields no pools; callers fall back to `allGpuPool`. + */ +export function tracePools(series: Pick): PowerPool[] { + const rows = new Map(); + series.devices?.forEach((device, row) => { + if (!device.role) return; + if (!rows.has(device.role)) rows.set(device.role, []); + rows.get(device.role)!.push(row); + }); + return POOL_ORDER.filter((role) => rows.has(role)).map((role) => ({ + role, + rows: rows.get(role)!, + })); +} + +export interface PoolSizeGroup { + size: number; + roles: PowerPoolRole[]; +} + +/** + * Pools that hold the same number of GPUs share one rated ceiling, so their + * reference draws once, labelled `prefill / decode ×16`, instead of two labels + * printed over each other. Sizes ascending, roles in pool order, deduplicated. + */ +export function groupPoolsBySize( + pools: readonly Pick[], +): PoolSizeGroup[] { + const bySize = new Map>(); + for (const pool of pools) { + if (!bySize.has(pool.rows.length)) bySize.set(pool.rows.length, new Set()); + bySize.get(pool.rows.length)!.add(pool.role); + } + return [...bySize.entries()] + .toSorted(([a], [b]) => a - b) + .map(([size, roles]) => ({ + size, + roles: POOL_ORDER.filter((role) => roles.has(role)), + })); +} + +/** + * Label row for each reference line: lines at the same watts (different + * hardware with an equal pool ceiling) stack their labels upward, slot 0 on + * the line and slot n `n` rows above, instead of overprinting. Input order. + */ +export function referenceLabelSlots(lines: readonly { watts: number }[]): number[] { + const used = new Map(); + return lines.map((line) => { + const slot = used.get(line.watts) ?? 0; + used.set(line.watts, slot + 1); + return slot; + }); +} + +/** Every GPU of the series as one pool. */ +export function allGpuPool(series: Pick): PowerPool { + return { role: 'all', rows: series.power.map((_, row) => row) }; +} + +// ── Deep link from a pinned scatter tooltip ───────────────────────────────── +// +// "View power trace" on a pinned tooltip switches the metric to the Timeline +// display; the timeline mounts afterwards and reads the requested trace here +// so it can emphasise that config. Module state rather than URL state: the +// focus is a one-shot gesture, and the share link stays `i_metric` alone. + +let pendingFocus: string | null = null; + +export function requestPowerTraceFocus(key: string): void { + pendingFocus = key; +} + +/** The pending focus request, cleared on read. */ +export function consumePowerTraceFocus(): string | null { + const key = pendingFocus; + pendingFocus = null; + return key; +} diff --git a/packages/app/src/components/inference/utils/role-energy.test.ts b/packages/app/src/components/inference/utils/role-energy.test.ts new file mode 100644 index 000000000..102703cc0 --- /dev/null +++ b/packages/app/src/components/inference/utils/role-energy.test.ts @@ -0,0 +1,26 @@ +import { describe, expect, it } from 'vitest'; + +import { reconstructedRoleEnergy } from './role-energy'; + +// A disaggregated 8K/1K run: the deployment's energy over the window divided +// by 8× more input than output tokens, so J/out ÷ J/in = 7.9 = the served ratio. +const entry = { + disagg: true, + power_valid: 1, + power_metric_schema_version: 2, + joules_per_input_token: 1, + joules_per_output_token: 7.9, + prefill_joules_per_input_token: 0.25, + decode_joules_per_output_token: 6, +}; + +describe('reconstructedRoleEnergy', () => { + it('expresses prefill energy per output token with the served token ratio and sums the roles', () => { + expect(reconstructedRoleEnergy(entry)).toEqual({ + prefill: 1.975, + decode: 6, + total: 7.975, + prefillShare: (100 * 1.975) / 7.975, + }); + }); +}); diff --git a/packages/app/src/components/inference/utils/role-energy.ts b/packages/app/src/components/inference/utils/role-energy.ts new file mode 100644 index 000000000..89e6b37f2 --- /dev/null +++ b/packages/app/src/components/inference/utils/role-energy.ts @@ -0,0 +1,64 @@ +/** + * Reconstructs how a disaggregated deployment's request energy splits between + * its prefill and decode pools (PowerX Figure 7). + * + * Schema-2 aggregate energy has one numerator: the deployment's energy over the + * validated window is divided by input tokens for `joules_per_input_token` and + * by output tokens for `joules_per_output_token`. Their ratio is therefore the + * input:output token ratio the benchmark actually served. Multiplying the + * prefill pool's J per input token by that ratio expresses the prefill energy + * per output token, on the same axis as the decode pool's J per output token. + * The two add up to the whole request's J per output token, which equals the + * deployment figure whenever the role energies partition the deployment energy. + * + * Nothing is estimated: every input is a same-window telemetry figure, and the + * result is `undefined` whenever one is missing, not validated, or not from a + * disaggregated deployment. + */ +export interface RoleEnergyInput { + disagg?: boolean; + power_valid?: number; + power_metric_schema_version?: number; + joules_per_input_token?: number; + joules_per_output_token?: number; + prefill_joules_per_input_token?: number; + decode_joules_per_output_token?: number; +} + +export interface ReconstructedRoleEnergy { + /** Prefill pool energy per output token (J). */ + prefill: number; + /** Decode pool energy per output token (J). */ + decode: number; + /** Prefill + decode (J per output token). */ + total: number; + /** Prefill share of the reconstructed total, in percent. */ + prefillShare: number; +} + +const positive = (value: unknown): value is number => + typeof value === 'number' && Number.isFinite(value) && value > 0; + +export function reconstructedRoleEnergy( + entry: RoleEnergyInput, +): ReconstructedRoleEnergy | undefined { + if (!entry.disagg || entry.power_valid !== 1 || entry.power_metric_schema_version !== 2) { + return undefined; + } + const input = entry.joules_per_input_token; + const output = entry.joules_per_output_token; + const prefill = entry.prefill_joules_per_input_token; + const decode = entry.decode_joules_per_output_token; + if (!positive(input) || !positive(output) || !positive(prefill) || !positive(decode)) { + return undefined; + } + const prefillPerOutputToken = prefill * (output / input); + const total = prefillPerOutputToken + decode; + if (!positive(prefillPerOutputToken) || !positive(total)) return undefined; + return { + prefill: prefillPerOutputToken, + decode, + total, + prefillShare: (100 * prefillPerOutputToken) / total, + }; +} diff --git a/packages/app/src/components/inference/utils/tooltip-utils.power-trace.test.ts b/packages/app/src/components/inference/utils/tooltip-utils.power-trace.test.ts new file mode 100644 index 000000000..45157d317 --- /dev/null +++ b/packages/app/src/components/inference/utils/tooltip-utils.power-trace.test.ts @@ -0,0 +1,94 @@ +import { afterEach, describe, expect, it, vi } from 'vitest'; + +import type { HardwareConfig, InferenceData, OverlayData } from '@/components/inference/types'; +import { + generateOverlayTooltipContent, + type OverlayTooltipConfig, + type TooltipConfig, +} from '@/components/inference/utils/tooltipUtils'; + +// "View power trace" on pinned tooltips: the deep link from a measured-power +// scatter point to its per-second telemetry on the Timeline display. + +const RUN_URL = 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34716669498'; +const AUDIT_NAME = + 'dsv4_8k1k_fp4_sglang_tp8-pp1-dcp1-pcp1-ep1-dpafalse_disagg-false_spec-none_conc64_b200-host-0123'; +const ACTION = 'data-action="view-power-trace"'; + +const hardwareConfig = { + b200: { + name: 'b200', + label: 'B200', + suffix: '', + gpu: 'B200', + color: 'blue', + power: 1000, + costh: 5, + costr: 1.25, + }, +} as unknown as HardwareConfig; + +function measuredPoint(overrides: Partial = {}): InferenceData { + return { + id: 980001, + date: '2026-09-01', + x: 60, + y: 600, + tp: 8, + conc: 64, + hwKey: 'b200', + precision: 'fp4', + benchmark_type: 'single_turn', + run_url: RUN_URL, + power_audit: { source: `power_validation_${AUDIT_NAME}.json` }, + tpPerGpu: { y: 400, roof: false }, + tpPerMw: { y: 50, roof: false }, + costh: { y: 1, roof: false }, + costr: { y: 1, roof: false }, + costhi: { y: 1, roof: false }, + costri: { y: 1, roof: false }, + ...overrides, + } as InferenceData; +} + +function config(overrides: Partial = {}): TooltipConfig { + return { + data: measuredPoint(), + isPinned: true, + xLabel: 'Interactivity (tok/s/user)', + yLabel: 'Measured Power per Chip (W)', + selectedYAxisMetric: 'y_measuredAvgPower', + hardwareConfig, + ...overrides, + }; +} + +function overlayConfig(overrides: Partial = {}): OverlayTooltipConfig { + return { + ...config({ data: measuredPoint({ id: 0 }) }), + overlayData: { + label: 'powerx-timeline', + hardwareConfig, + data: [], + runUrl: RUN_URL, + } as unknown as OverlayData, + ...overrides, + }; +} + +describe('View power trace tooltip action', () => { + afterEach(() => { + vi.unstubAllGlobals(); + }); + + it('renders on pinned overlay tooltips whose points carry id 0', () => { + const html = generateOverlayTooltipContent(overlayConfig()); + expect(html).toContain(` { + const variant = d.powerVariant; + if (!variant) return ''; + const t = TOOLTIP_STRINGS[locale]; + let html = tooltipLine(t.series, powerVariantLabel(variant, locale)); + if ( + variant.kind === 'role' && + variant.id !== 'all' && + getMeasuredMetricConfig(selectedYAxisMetric)?.family === 'energy' + ) { + const energy = reconstructedRoleEnergy(d); + if (energy) { + const share = variant.id === 'prefill' ? energy.prefillShare : 100 - energy.prefillShare; + html += tooltipLine(t.roleEnergyShare, `${share.toFixed(1)}%`); + } + } + return html; +}; + const totalChipsHTML = (d: InferenceData, selectedYAxisMetric: string, locale: Locale): string => { const t = TOOLTIP_STRINGS[locale]; const { physical, configured } = chipCounts( @@ -228,6 +270,8 @@ const SYSTEM_POWER_STRINGS = { : `${chassis} eight-GPU chassis · ${measured} of ${modeled} GPUs measured, extrapolated to full chassis`, extrapolation: 'Unmeasured chassis GPUs are assumed to run the same workload at the measured per-GPU power; deployment values are the measured GPUs’ share.', + uniformHosts: + 'No per-host telemetry for this multinode deployment; every chassis is modeled at the deployment-mean GPU power.', normalization: 'AC power is divided by all modeled chassis GPUs, including prefill and decode.', boundary: 'Includes GPU chassis CPUs; excludes separate CPU-only frontend/router hosts.', model: 'Power model source', @@ -257,6 +301,7 @@ const SYSTEM_POWER_STRINGS = { : `${chassis} 个八卡机箱 · 实测 ${measured}/${modeled} 张 GPU,按满机箱外推`, extrapolation: '假设机箱内未实测的 GPU 运行相同负载、功耗与实测每卡功耗相同;部署数值为实测 GPU 所占份额。', + uniformHosts: '该多节点部署没有逐主机功耗数据;每个机箱按部署平均每卡功耗建模。', normalization: '交流功耗按所有建模机箱的 GPU 总数分摊,包括 Prefill 与 Decode。', boundary: '计入 GPU 机箱内的 CPU;不计入独立的纯 CPU 前端或路由主机。', model: '功耗模型来源', @@ -303,7 +348,7 @@ const modeledSystemPowerHTML = ( ? ` ${tooltipLine(t.deploymentAc, `${fmt(estimate.deploymentAcWatts)} W`)} ${tooltipLine(`${t.facility} (PUE ${fmt(estimate.pue)})`, `${fmt(estimate.deploymentFacilityWatts)} W`)} -
    ${t.topology(estimate.chassisCount, estimate.gpuCount, estimate.modeledGpuCount)}${estimate.chassisBasis === 'extrapolated' ? `
    ${t.extrapolation}` : ''}
    ${t.assumptions}
    ${t.platformAssumptions}
    ${t.normalization}
    ${t.boundary}
    +
    ${t.topology(estimate.chassisCount, estimate.gpuCount, estimate.modeledGpuCount)}${estimate.chassisBasis === 'extrapolated' ? `
    ${t.extrapolation}` : ''}${estimate.topologyBasis === 'uniform-hosts' ? `
    ${t.uniformHosts}` : ''}
    ${t.assumptions}
    ${t.platformAssumptions}
    ${t.normalization}
    ${t.boundary}
    ${tooltipLine(t.model, `
    ${escapeHtml(estimate.hardware)} · ${escapeHtml(estimate.modelRevision.slice(0, 12))}`)} ${t.sweep} ` @@ -493,39 +538,108 @@ const generateAgenticHTML = (d: InferenceData, locale: Locale): string => { }; const ACTION_STRINGS = { - en: { charts: 'View charts', logs: 'View logs' }, - zh: { charts: '查看图表', logs: '查看日志' }, + en: { + charts: 'View charts', + logs: 'View logs', + powerTelemetry: 'View PowerX', + powerTrace: 'View power trace', + }, + zh: { + charts: '查看图表', + logs: '查看日志', + powerTelemetry: '查看 PowerX', + powerTrace: '查看功耗曲线', + }, } as const; -const pointDetailActionLink = (action: 'view-charts' | 'view-logs', href: string, label: string) => +type TooltipAction = 'view-charts' | 'view-logs' | 'view-power-trace'; + +const pointDetailActionLink = (action: TooltipAction, href: string, label: string) => `${label} →`; -/** Point-detail links rendered only for persisted, pinned official points. */ -const viewActionsHTML = ( - isPinned: boolean, - hasTraceData: boolean, - hasLogData: boolean, - pointId: number | undefined, - benchmarkType: string | undefined, - locale: Locale, -): string => { - const isAgentic = benchmarkType === 'agentic_traces'; - const showCharts = isAgentic && hasTraceData; - if (!isPinned || !isPersistedBenchmarkId(pointId) || (!showCharts && !hasLogData)) return ''; - const prefix = locale === 'zh' ? '/zh' : ''; - const agenticHref = agenticDetailHref(pointId, locale); - const logHref = isAgentic - ? `${agenticHref}${agenticHref.includes('?') ? '&' : '?'}view=logs` - : `${prefix}/inference/logs/${pointId}`; +/** + * Whether a point on the measured-power / energy scatter can jump to its + * per-second telemetry on the Timeline display. Overlay points qualify too: + * the trace is keyed by run id and audit name, not by a persisted row id. + */ +export const showsPowerTraceAction = ( + point: Pick, + selectedYAxisMetric: string, +): boolean => + selectedYAxisMetric !== POWER_TIMELINE_METRIC_KEY && + getMeasuredMetricConfig(selectedYAxisMetric) !== undefined && + traceKeyForPoint(point) !== null; + +/** + * Same-tab click is intercepted by the chart (in-page metric switch); the href + * keeps open-in-new-tab landing on the Timeline display of THIS chart. The + * address bar is stripped of chart state after load, so the share-link store + * is layered over the live location first (`chartStateHref`). + */ +const powerTraceHref = (): string => + typeof window === 'undefined' ? '#' : chartStateHref({ i_metric: POWER_TIMELINE_METRIC_KEY }); + +interface ViewActionsInput { + isPinned: boolean; + hasTraceData: boolean; + hasLogData: boolean; + point: InferenceData; + /** + * Metric the chart currently plots, when that chart can switch to the + * Timeline display in place (the scatter). Omitted by charts that cannot, + * so they never render a "View power trace" link nothing would handle. + */ + powerTraceMetric?: string; + showPowerTelemetry?: boolean; + locale: Locale; +} + +/** + * Point-detail links rendered only on pinned tooltips. "View charts" and + * "View logs" need a persisted row id (overlay points have none); "View power + * trace" needs only the run and audit source, so it works for overlays too. + */ +const viewActionsHTML = ({ + isPinned, + hasTraceData, + hasLogData, + point, + powerTraceMetric, + showPowerTelemetry, + locale, +}: ViewActionsInput): string => { + if (!isPinned) return ''; const t = ACTION_STRINGS[locale]; - const actions = [ - showCharts ? pointDetailActionLink('view-charts', agenticHref, t.charts) : '', - hasLogData ? pointDetailActionLink('view-logs', logHref, t.logs) : '', - ].filter(Boolean); + const actions: string[] = []; + const pointId = point.id; + const isAgentic = point.benchmark_type === 'agentic_traces'; + const showCharts = isAgentic && hasTraceData; + if (isPersistedBenchmarkId(pointId) && (showCharts || hasLogData)) { + const prefix = locale === 'zh' ? '/zh' : ''; + const agenticHref = agenticDetailHref(pointId, locale); + if (showCharts) { + actions.push(pointDetailActionLink('view-charts', agenticHref, t.charts)); + } + if (hasLogData) { + const logHref = isAgentic + ? `${agenticHref}${agenticHref.includes('?') ? '&' : '?'}view=logs` + : `${prefix}/inference/logs/${pointId}`; + actions.push(pointDetailActionLink('view-logs', logHref, t.logs)); + } + } + if (showPowerTelemetry && isPersistedBenchmarkId(pointId)) { + actions.push( + ``, + ); + } + if (powerTraceMetric !== undefined && showsPowerTraceAction(point, powerTraceMetric)) { + actions.push(pointDetailActionLink('view-power-trace', powerTraceHref(), t.powerTrace)); + } + if (actions.length === 0) return ''; return `
    ${actions.join('')}
    `; }; @@ -708,6 +822,7 @@ export const generateTooltipContent = (config: TooltipConfig): string => { : '' } ${powerTierHTML(d, selectedYAxisMetric, locale)} + ${powerVariantHTML(d, selectedYAxisMetric, locale)} ${modeledSystemPowerHTML(d, selectedYAxisMetric, isPinned, locale)} ${totalChipsHTML(d, selectedYAxisMetric, locale)} ${generateParallelismHTML(d, locale)} @@ -718,7 +833,15 @@ export const generateTooltipContent = (config: TooltipConfig): string => { ${generateAgenticHTML(d, locale)} ${generateWorkerPowerHTML(d, isPinned, locale)} ${runLinkHTML(runUrl, locale)} - ${viewActionsHTML(isPinned, Boolean(hasTrace), Boolean(config.hasLog), d.id, d.benchmark_type, locale)} + ${viewActionsHTML({ + isPinned, + hasTraceData: Boolean(hasTrace), + hasLogData: Boolean(config.hasLog), + point: d, + showPowerTelemetry: config.showPowerTelemetry, + powerTraceMetric: selectedYAxisMetric, + locale, + })} `; }; @@ -752,6 +875,7 @@ export const generateOverlayTooltipContent = (config: OverlayTooltipConfig): str ${tooltipLine(xLabel, fmt(d.x))} ${tooltipLine(yLabel, fmt(d.y))} ${powerTierHTML(d, selectedYAxisMetric, locale)} + ${powerVariantHTML(d, selectedYAxisMetric, locale)} ${modeledSystemPowerHTML(d, selectedYAxisMetric, isPinned, locale)} ${totalChipsHTML(d, selectedYAxisMetric, locale)} ${generateParallelismHTML(d, locale)} @@ -761,6 +885,14 @@ export const generateOverlayTooltipContent = (config: OverlayTooltipConfig): str ${powerWithheldHTML(d, locale)} ${generateAgenticHTML(d, locale)} ${generateWorkerPowerHTML(d, isPinned, locale)} + ${viewActionsHTML({ + isPinned, + hasTraceData: false, + hasLogData: false, + point: d, + powerTraceMetric: selectedYAxisMetric, + locale, + })} `; }; @@ -813,6 +945,7 @@ export const generateGPUGraphTooltipContent = (config: TooltipConfig): string => : '' } ${powerTierHTML(d, selectedYAxisMetric, locale)} + ${powerVariantHTML(d, selectedYAxisMetric, locale)} ${modeledSystemPowerHTML(d, selectedYAxisMetric, isPinned, locale)} ${totalChipsHTML(d, selectedYAxisMetric, locale)} ${generateParallelismHTML(d, locale)} @@ -823,7 +956,14 @@ export const generateGPUGraphTooltipContent = (config: TooltipConfig): string => ${generateAgenticHTML(d, locale)} ${generateWorkerPowerHTML(d, isPinned, locale)} ${runLinkHTML(runUrl, locale)} - ${viewActionsHTML(isPinned, Boolean(hasTrace), Boolean(hasLog), d.id, d.benchmark_type, locale)} + ${viewActionsHTML({ + isPinned, + hasTraceData: Boolean(hasTrace), + hasLogData: Boolean(hasLog), + point: d, + showPowerTelemetry: config.showPowerTelemetry, + locale, + })} `; }; diff --git a/packages/app/src/lib/chart-utils.test.ts b/packages/app/src/lib/chart-utils.test.ts index 9d5203b66..a23699602 100644 --- a/packages/app/src/lib/chart-utils.test.ts +++ b/packages/app/src/lib/chart-utils.test.ts @@ -928,6 +928,89 @@ describe('createChartDataPoint energy fields', () => { }); }); +// =========================================================================== +// createChartDataPoint — power-boundary fields (B2 GPU provisioned, B3 utility +// provisioned, B4 utility modeled). Mock specs: tdp 700 W, power 700 "kW". +// =========================================================================== +const boundaryPoint = (e: AggDataEntry) => + createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); + +describe('createChartDataPoint power-boundary fields', () => { + // Eight measured GPUs (2 prefill + 6 decode) on two partially filled chassis: + // the model evaluates 16 GPUs, so only deploymentFacilityWatts ÷ gpuCount yields 877.5 W. + // Every other numerator/denominator pairing gives a different number. + const supportedModel = { + status: 'supported' as const, + hardware: 'h100', + modelRevision: 'test', + modelPath: 'test', + gpuCount: 8, + chassisCount: 2, + modeledGpuCount: 16, + measuredGpuWattsPerGpu: 500, + chassisAcWatts: 10_400, + chassisAcWattsPerGpu: 650, + facilityWatts: 13_520, + deploymentAcWatts: 5400, + deploymentFacilityWatts: 7020, + pue: 1.3, + telemetryBasis: 'validated-v2' as const, + topologyBasis: 'worker-hosts' as const, + chassisBasis: 'extrapolated' as const, + }; + const validated = { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 500, + joules_per_output_token: 10, + modeledSystemPower: supportedModel, + }; + it('emits all six boundary fields for a validated official row', () => { + const p = boundaryPoint( + entry({ output_tput_per_gpu: 400, benchmark_type: 'single_turn', ...validated }), + ); + expect(p.gpuProvisionedWatts).toEqual({ y: 700, roof: false }); + expect(p.gpuProvisionedJPerOutputToken?.y).toBeCloseTo(700 / 400, 10); + expect(p.utilityProvisionedWatts).toEqual({ y: 700_000, roof: false }); + expect(p.utilityProvisionedJPerOutputToken?.y).toBeCloseTo(700_000 / 400, 10); + // B4 W = deployment facility ÷ measured GPUs; J scales B1 by B4 W ÷ B1 W. + expect(p.utilityModeledWatts).toEqual({ y: 877.5, roof: false }); + expect(p.utilityModeledJPerOutputToken?.y).toBeCloseTo((10 * 877.5) / 500, 10); + // Existing measured (B1) fields are untouched by the new boundaries. + expect(p.measuredAvgPower).toEqual({ y: 500, roof: false }); + expect(p.measuredJPerOutputToken).toEqual({ y: 10, roof: false }); + }); + + it('normalizes fixed-sequence disaggregated energy by all GPUs while jOutput stays per decode GPU', () => { + const p = boundaryPoint( + entry({ + output_tput_per_gpu: 400, + disagg: true, + benchmark_type: 'single_turn', + num_prefill_gpu: 4, + num_decode_gpu: 4, + }), + ); + expect(p.gpuProvisionedJPerOutputToken?.y).toBeCloseTo((700 * 8) / (400 * 4), 10); + expect(p.utilityProvisionedJPerOutputToken?.y).toBeCloseTo((700_000 * 8) / (400 * 4), 10); + expect(p.jOutput?.y).toBeCloseTo(700_000 / 400, 10); + expect(p.utilityProvisionedJPerOutputToken?.y).toBeCloseTo(2 * p.jOutput!.y, 10); + + const agentic = boundaryPoint( + entry({ + output_tput_per_gpu: 400, + disagg: true, + benchmark_type: 'agentic_traces', + num_prefill_gpu: 4, + num_decode_gpu: 4, + }), + ); + expect(agentic.gpuProvisionedWatts?.y).toBe(700); + expect(agentic.gpuProvisionedJPerOutputToken).toBeUndefined(); + expect(agentic.utilityProvisionedJPerOutputToken).toBeUndefined(); + }); +}); + // =========================================================================== // createChartDataPoint — measured power / energy fields (from runner telemetry) // =========================================================================== @@ -952,20 +1035,6 @@ describe('createChartDataPoint measured power fields', () => { expect(missing.measuredP75Power).toBeUndefined(); expect(missing.measuredP90Power).toBeUndefined(); }); - it('emits measuredAvgPower when avg_power_w is present on the entry', () => { - const e = entry({ avg_power_w: 685.5 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredAvgPower).toBeDefined(); - expect(point.measuredAvgPower!.y).toBe(685.5); - expect(point.measuredAvgPower!.roof).toBe(false); - }); - - it('emits measuredJPerOutputToken when joules_per_output_token is present', () => { - const e = entry({ joules_per_output_token: 8.4 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredJPerOutputToken).toBeDefined(); - expect(point.measuredJPerOutputToken!.y).toBe(8.4); - }); it('derives J/query, Wh/query, and percent TDP from validated source fields', () => { const e = entry({ avg_power_w: 560, joules_per_successful_query: 1800 }); @@ -1019,14 +1088,6 @@ describe('createChartDataPoint measured power fields', () => { expect(point.measuredAvgPower!.y).toBe(0); }); - it('emits measuredJPerTotalToken when joules_per_total_token is present', () => { - const e = entry({ joules_per_total_token: 0.93 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredJPerTotalToken).toBeDefined(); - expect(point.measuredJPerTotalToken!.y).toBe(0.93); - expect(point.measuredJPerTotalToken!.roof).toBe(false); - }); - it('emits J/output and J/total independently — different denominators', () => { // 8k1k workload: J/output ≈ 9 × J/total (input is ~8x output, so output/total ≈ 1/9). const e = entry({ joules_per_output_token: 2.04, joules_per_total_token: 0.23 }); @@ -1051,30 +1112,6 @@ describe('createChartDataPoint measured power fields', () => { // createChartDataPoint — per-stage measured power / energy (disagg prefill/decode) // =========================================================================== describe('createChartDataPoint per-stage measured power fields', () => { - it('emits measuredPrefillAvgPower when prefill_avg_power_w is present', () => { - const e = entry({ prefill_avg_power_w: 920.3 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredPrefillAvgPower).toBeDefined(); - expect(point.measuredPrefillAvgPower!.y).toBe(920.3); - expect(point.measuredPrefillAvgPower!.roof).toBe(false); - }); - - it('emits measuredDecodeAvgPower when decode_avg_power_w is present', () => { - const e = entry({ decode_avg_power_w: 612.1 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredDecodeAvgPower).toBeDefined(); - expect(point.measuredDecodeAvgPower!.y).toBe(612.1); - expect(point.measuredDecodeAvgPower!.roof).toBe(false); - }); - - it('emits measuredJPerInputToken when joules_per_input_token is present', () => { - const e = entry({ joules_per_input_token: 0.27 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredJPerInputToken).toBeDefined(); - expect(point.measuredJPerInputToken!.y).toBe(0.27); - expect(point.measuredJPerInputToken!.roof).toBe(false); - }); - it('omits all per-stage fields on legacy rows predating per-stage attribution', () => { // Single-node / pre-disagg runs emit avg_power_w only, no prefill/decode split. const e = entry({ avg_power_w: 685.5 }); @@ -1084,15 +1121,6 @@ describe('createChartDataPoint per-stage measured power fields', () => { expect(point.measuredJPerInputToken).toBeUndefined(); }); - it('emits prefill and decode independently — the disagg per-stage split', () => { - // GB300 disagg: prefill GPUs run compute-bound (higher W) than decode GPUs. - const e = entry({ prefill_avg_power_w: 948, decode_avg_power_w: 631 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredPrefillAvgPower!.y).toBe(948); - expect(point.measuredDecodeAvgPower!.y).toBe(631); - expect(point.measuredPrefillAvgPower!.y).toBeGreaterThan(point.measuredDecodeAvgPower!.y); - }); - it('preserves a zero per-stage power value (not falsy-coerced away)', () => { // Same typeof===number gate as total power — 0 W must survive, not be dropped. const e = entry({ prefill_avg_power_w: 0, decode_avg_power_w: 0 }); diff --git a/packages/app/src/lib/chart-utils.ts b/packages/app/src/lib/chart-utils.ts index e94032342..78b62b2d8 100644 --- a/packages/app/src/lib/chart-utils.ts +++ b/packages/app/src/lib/chart-utils.ts @@ -11,6 +11,7 @@ import type { AggDataEntry, ChartDefinition, InferenceData, + PowerBasisFieldKey, YAxisMetricKey, } from '@/components/inference/types'; import { @@ -20,6 +21,8 @@ import { import { DEFAULT_TCO_BASIS, getGpuSpecs, isKnownGpu, type TcoBasis } from '@/lib/constants'; import { getVendor, type Vendor } from '@/lib/dynamic-colors'; import type { Locale } from '@/lib/i18n'; +import { buildPowerBasisChartFields, type PowerBasisChartFields } from '@/lib/power-basis'; +import { reconstructedRoleEnergy } from '@/components/inference/utils/role-energy'; // --------------------------------------------------------------------------- // High-contrast color generation (iwanthue — k-means in CIELab) @@ -289,7 +292,13 @@ export function buildAvailabilityHwKey( return hwKey; } -export type DerivedMetricKey = BenchmarkMetricKey; +// Power-boundary fields are derived here before the registry exposes them as +// axes; the union collapses once METRIC_REGISTRY carries the same keys. The +// reconstructed prefill energy is a comparison-only series (never an axis). +export type DerivedMetricKey = + | BenchmarkMetricKey + | PowerBasisFieldKey + | 'reconstructedPrefillJPerOutputToken'; export type DerivedChartFields = Pick; const chartMetric = (y: number): { y: number; roof: boolean } => ({ y, roof: false }); @@ -411,6 +420,11 @@ export function buildDerivedChartFields( hardwarePower && tputPerGpu ? (hardwarePower * 1000) / tputPerGpu : 0, ); } + // jOutput keeps the historical per-GPU normalization: for disaggregated rows + // output_tput_per_gpu is per decode GPU, so this is all-in W of one decode GPU + // per output token and ignores the prefill pool. The power-boundary field + // utilityProvisionedJPerOutputToken uses the same all-in W but counts every + // allocated GPU, so the two differ on disaggregated rows by (P + D) / D. if (hardwarePower > 0 && wants('jOutput') && outputTputPerGpu) { fields.jOutput = chartMetric(hardwarePower ? (hardwarePower * 1000) / outputTputPerGpu : 0); } @@ -426,6 +440,14 @@ export function buildDerivedChartFields( if (wants(key)) fields[key] = value; } + const powerBasis = buildPowerBasisChartFields(entry, specs); + for (const [key, value] of Object.entries(powerBasis) as [ + keyof PowerBasisChartFields, + { y: number; roof: boolean }, + ][]) { + if (wants(key)) fields[key] = value; + } + if (wants('modeledChassisPowerPerGpu') && entry.modeledSystemPower?.status === 'supported') { fields.modeledChassisPowerPerGpu = chartMetric(entry.modeledSystemPower.chassisAcWattsPerGpu); } @@ -521,6 +543,8 @@ type MeasuredPowerChartFields = Partial< | 'measuredJPerSuccessfulQuery' | 'measuredWhPerSuccessfulQuery' | 'measuredPowerPercentTdp' + | 'measuredPowerTimeline' + | 'reconstructedPrefillJPerOutputToken' > >; @@ -530,8 +554,13 @@ function buildMeasuredPowerChartFields( tdpWatts: number, ): MeasuredPowerChartFields { return { + // The timeline axis aliases the validated average: the point set (and + // its table row) is the same, only the chart body changes. ...(typeof entry.avg_power_w === 'number' - ? { measuredAvgPower: chartMetric(entry.avg_power_w) } + ? { + measuredAvgPower: chartMetric(entry.avg_power_w), + measuredPowerTimeline: chartMetric(entry.avg_power_w), + } : {}), ...(typeof entry.p75_power_w === 'number' && Number.isFinite(entry.p75_power_w) ? { measuredP75Power: chartMetric(entry.p75_power_w) } @@ -562,6 +591,14 @@ function buildMeasuredPowerChartFields( ...(typeof entry.decode_joules_per_output_token === 'number' ? { measuredDecodeJPerOutputToken: chartMetric(entry.decode_joules_per_output_token) } : {}), + // Prefill energy on the output-token axis, so the roles comparison can + // stack it against the decode pool (PowerX Figure 7). + ...(() => { + const roleEnergy = reconstructedRoleEnergy(entry); + return roleEnergy + ? { reconstructedPrefillJPerOutputToken: chartMetric(roleEnergy.prefill) } + : {}; + })(), ...(typeof entry.joules_per_successful_query === 'number' ? { measuredJPerSuccessfulQuery: chartMetric(entry.joules_per_successful_query), diff --git a/packages/app/src/lib/csv-export-helpers.test.ts b/packages/app/src/lib/csv-export-helpers.test.ts index 5fe616931..316f6af1f 100644 --- a/packages/app/src/lib/csv-export-helpers.test.ts +++ b/packages/app/src/lib/csv-export-helpers.test.ts @@ -8,6 +8,7 @@ import { historicalTrendToCsv, } from './csv-export-helpers'; import type { InferenceData } from '@/components/inference/types'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; const makePoint = (overrides: Partial = {}): InferenceData => ({ x: 100, @@ -752,3 +753,39 @@ describe('historicalTrendToCsv (mirrors HistoricalTrendsDisplay export)', () => expect(rows[1][headers.indexOf('Date')]).toBe('2025-01-15'); }); }); + +describe('inferenceChartToCsv power comparison', () => { + it.each([false, true])('exports plotted role values with overlay=%s', (overlay) => { + const base = makePoint({ + hwKey: 'gb300_dynamo-trt', + y: 708.1, + measuredAvgPower: { y: 708.1, roof: false }, + measuredPrefillAvgPower: { y: 760.442, roof: false }, + measuredDecodeAvgPower: { y: 690.652, roof: false }, + run_url: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35532106109', + }); + const points = expandPowerCompareSeries([base], 'y_measuredAvgPower', 'roles'); + const { headers, rows } = inferenceChartToCsv( + overlay ? [] : points, + 'Kimi-K3', + 'agentic-traces', + overlay ? points : [], + { + yHeader: 'Measured Power per Chip (W)', + yPath: 'measuredAvgPower.y', + xHeader: 'Interactivity (tok/s/user)', + }, + ); + expect( + rows.map((row) => [ + row[headers.indexOf('Power Series')], + row[headers.indexOf('Measured Power per Chip (W)')], + ]), + ).toEqual([ + ['All GPUs', 708.1], + ['Prefill GPUs', 760.442], + ['Decode GPUs', 690.652], + ]); + expect(points.every((point) => point.measuredAvgPower?.y === 708.1)).toBe(true); + }); +}); diff --git a/packages/app/src/lib/csv-export-helpers.ts b/packages/app/src/lib/csv-export-helpers.ts index 576569af8..a58ba948f 100644 --- a/packages/app/src/lib/csv-export-helpers.ts +++ b/packages/app/src/lib/csv-export-helpers.ts @@ -9,6 +9,7 @@ import { METRIC_REGISTRY } from '@/components/inference/metric-registry'; import type { InferenceData, TrendDataPoint } from '@/components/inference/types'; +import { inferPowerCompare, powerSeriesLabel } from '@/components/inference/utils/power-compare'; import { chipCounts } from '@/lib/chip-counts'; import type { SubmissionVolumeRow } from '@/lib/submissions-types'; @@ -57,6 +58,12 @@ export function inferenceChartToCsv( const islOsl = sequenceToIslOsl(sequence); const showModeledPower = displayedMetrics?.yPath === METRIC_REGISTRY.modeledChassisPowerPerGpu.field; + // A power comparison (`i_pcompare`) appends boundary / role clones of the + // plotted points; name each row's series so the export stays unambiguous. + const allPoints = [...data, ...overlayData]; + const powerCompare = inferPowerCompare(allPoints); + const showPowerSeries = powerCompare !== 'none'; + const plottedMetric = displayedMetrics ? `y_${displayedMetrics.yPath.split('.')[0]}` : ''; const headers = [ 'Model', 'ISL', @@ -111,13 +118,15 @@ export function inferenceChartToCsv( 'Physical Chips', 'DP', ...(showModeledPower ? ['Configured Chip Count'] : []), + ...(showPowerSeries ? ['Power Series'] : []), ]; const displayedColumns = displayedMetrics ? [ { header: displayedMetrics.yHeader, - value: (point: InferenceData) => nestedMetric(point, displayedMetrics.yPath), + value: (point: InferenceData) => + point.powerVariant ? point.y : nestedMetric(point, displayedMetrics.yPath), }, { header: displayedMetrics.xHeader, value: (point: InferenceData) => point.x }, ].filter( @@ -128,7 +137,7 @@ export function inferenceChartToCsv( : []; headers.splice(10, 0, ...displayedColumns.map((column) => column.header)); - const rows = [...data, ...overlayData] + const rows = allPoints .filter((d) => !d.hidden) .map((d) => { const chips = chipCounts(d, showModeledPower); @@ -177,6 +186,7 @@ export function inferenceChartToCsv( chips.physical, d.dp ?? '', ...(showModeledPower ? [chips.configured] : []), + ...(showPowerSeries ? [powerSeriesLabel(d, plottedMetric, powerCompare, 'en')] : []), ]; row.splice(10, 0, ...displayedColumns.map((column) => column.value(d))); return row; diff --git a/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts b/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts index d50fe1573..c7cc0ad80 100644 --- a/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts +++ b/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts @@ -16,10 +16,12 @@ import { isPerfRulerCurveVisible, movePerfRulerIsoX, nextPerfRulerState, + parsePerfRulers, pathXExtent, perfRulerCurveSet, prunePerfRulers, renderPerfRulers, + serializePerfRulers, type PerfRulerEndInput, type PerfRulerGeometry, type PerfRulerLabelLayoutOptions, @@ -889,6 +891,34 @@ describe('perfRulerCurveSet', () => { }); }); +// ── serializePerfRulers / parsePerfRulers (share links) ───────────── + +describe('serializePerfRulers / parsePerfRulers', () => { + const OFFICIAL_A = 'roofline-b200_trt_fp8'; + const OFFICIAL_B = 'roofline-mi355x_sglang_fp4'; + // Overlay curves carry the unofficial run index; power-envelope curves are + // split per date with the encoded date appended (`%2F` from a slash). + const OVERLAY = 'overlay-roofline-h100_vllm_fp8_run1__2026-09%2F11'; + + it('round-trips completed rulers, including overlay and date-scoped curve ids', () => { + const state = complete( + complete(EMPTY_PERF_RULER_STATE, OFFICIAL_A, OFFICIAL_B, 41.5), + OFFICIAL_A, + OVERLAY, + 120, + ); + const encoded = serializePerfRulers(state); + expect(encoded).toBe(`41.5|${OFFICIAL_A}|${OFFICIAL_B};120|${OFFICIAL_A}|${OVERLAY}`); + const parsed = parsePerfRulers(encoded); + expect(parsed.rulers).toEqual([ + { id: 1, curveA: OFFICIAL_A, curveB: OFFICIAL_B, isoX: 41.5 }, + { id: 2, curveA: OFFICIAL_A, curveB: OVERLAY, isoX: 120 }, + ]); + expect(parsed.draft).toBeNull(); + expect(parsed.nextId).toBe(3); + }); +}); + // ── pathXExtent ───────────────────────────────────────────── describe('pathXExtent', () => { diff --git a/packages/app/src/lib/d3-chart/layers/perf-ruler.ts b/packages/app/src/lib/d3-chart/layers/perf-ruler.ts index c44143529..ef13b2a1d 100644 --- a/packages/app/src/lib/d3-chart/layers/perf-ruler.ts +++ b/packages/app/src/lib/d3-chart/layers/perf-ruler.ts @@ -323,6 +323,72 @@ export function prunePerfRulers( return { ...prev, rulers, draft }; } +const PERF_RULER_URL_RULER_SEPARATOR = ';'; +const PERF_RULER_URL_FIELD_SEPARATOR = '|'; +/** + * Shape of a curve id the link may reference: one roofline path's identity + * class (`roofline-` / `overlay-roofline-`), never the shared + * `roofline-path` / `overlay-roofline-path` marker classes or any other node + * inside the zoom group — those match many paths, so a hand-edited link + * would draw a ruler between whichever two come first in DOM order. + */ +const PERF_RULER_CURVE_ID = /^(?:overlay-)?roofline-(?!path$)[\w%.-]+$/u; + +/** + * Share-link encoding of the COMPLETED rulers (`i_rulers`). One ruler per + * `;`, fields joined by `|`: `isoX|curveA|curveB`. Curve ids are the rendered + * roofline path identity classes (`roofline-_`, + * `overlay-roofline-__run`, optionally `__`), whose alphabet is `[A-Za-z0-9_%.-]`, so neither separator can + * appear inside one; `URLSearchParams` percent-encodes both on the wire. + * The iso-x is rounded to four significant digits to keep links short — a + * 0.05% shift on the x metric is far below the ruler's visual resolution. + * The draft is never serialized: it is an unfinished click, not a + * measurement. Empty state serializes to '' so the param strips as default. + */ +export function serializePerfRulers(state: PerfRulerState): string { + return state.rulers + .map((ruler) => + [Number(ruler.isoX.toPrecision(4)), ruler.curveA, ruler.curveB].join( + PERF_RULER_URL_FIELD_SEPARATOR, + ), + ) + .join(PERF_RULER_URL_RULER_SEPARATOR); +} + +/** + * Inverse of {@link serializePerfRulers}. Malformed entries (wrong field + * count, non-numeric iso-x, identical curve ids, or ids that are not + * roofline identity classes) are dropped silently — a hand-edited or + * truncated link degrades to fewer rulers, never to an error. The list is capped at + * {@link MAX_PERF_RULERS} keeping the NEWEST (last-serialized) entries, the + * same end the click reducer drops from. Ids are reassigned 1..n with + * `nextId = n + 1`, so parsed rulers are valid D3 join keys and a ruler + * placed afterwards never collides. Returns {@link EMPTY_PERF_RULER_STATE} + * (same reference) for '', null, or an all-malformed value. + */ +export function parsePerfRulers(raw: string | null | undefined): PerfRulerState { + if (!raw) return EMPTY_PERF_RULER_STATE; + const parsed: Omit[] = []; + for (const entry of raw.split(PERF_RULER_URL_RULER_SEPARATOR)) { + const fields = entry.split(PERF_RULER_URL_FIELD_SEPARATOR); + if (fields.length !== 3) continue; + const [isoXField, curveA, curveB] = fields; + if (isoXField.trim() === '' || curveA === curveB) continue; + if (!PERF_RULER_CURVE_ID.test(curveA) || !PERF_RULER_CURVE_ID.test(curveB)) continue; + const isoX = Number(isoXField); + if (!Number.isFinite(isoX)) continue; + parsed.push({ curveA, curveB, isoX }); + } + if (parsed.length === 0) return EMPTY_PERF_RULER_STATE; + const kept = parsed.slice(-MAX_PERF_RULERS); + return { + rulers: kept.map((ruler, index) => ({ id: index + 1, ...ruler })), + draft: null, + nextId: kept.length + 1, + }; +} + /** Every curve referenced by any ruler or the draft (hit-halo styling). */ export function perfRulerCurveSet(state: PerfRulerState): Set { const curves = new Set(); diff --git a/packages/app/src/lib/inference-labels.ts b/packages/app/src/lib/inference-labels.ts index c95883095..7854d7d41 100644 --- a/packages/app/src/lib/inference-labels.ts +++ b/packages/app/src/lib/inference-labels.ts @@ -41,6 +41,48 @@ export function inferenceFrameworkLabelOverride( } /** Keep unofficial-run identity/markers while making its special engine visible. */ +/** Leads every unofficial-run label, in line labels and legend rows alike. */ +export const OVERLAY_LABEL_MARKER = '✕ '; +const RUN_TAG_MAX = 20; +const RUN_TAG_TAIL = 17; + +export interface OverlayRunIdentity { + id: number | string; + branch?: string | null; +} + +/** + * Short, still recognisable name for a run: the whole branch when it is short, + * else its last path segment, else the branch tail (klaud nightlies end in + * `-`). Falls back to the run id when the branch is unknown. + */ +export function shortRunTag(run: OverlayRunIdentity): string { + const branch = run.branch?.trim() || `run ${run.id}`; + if (branch.length <= RUN_TAG_MAX) return branch; + const segment = branch.slice(branch.lastIndexOf('/') + 1); + if (segment.length > 0 && segment.length <= RUN_TAG_MAX) return segment; + return `…${branch.slice(-RUN_TAG_TAIL)}`; +} + +/** ` · ` appended to an overlay line label when other runs draw the same hardware. */ +export function overlayRunTag(run: OverlayRunIdentity): string { + return ` · ${shortRunTag(run)}`; +} + +/** + * Line-label text for an unofficial-run curve. Pills name the hardware, not the + * branch: branch names run to 70+ characters and the legend already carries + * them. The run tag is added only when several overlay runs draw the same + * hardware, so the pills stay distinguishable. + */ +export function getOverlayLineLabel( + hardwareLabel: string, + run: OverlayRunIdentity, + sharesHardware: boolean, +): string { + return `${OVERLAY_LABEL_MARKER}${hardwareLabel}${sharesHardware ? overlayRunTag(run) : ''}`; +} + export function getInferenceRunLabel( label: string, points: readonly (RunProvenance & { framework?: string })[], diff --git a/packages/app/src/lib/modeled-system-power.test.ts b/packages/app/src/lib/modeled-system-power.test.ts index c968537ae..c8fb4c26e 100644 --- a/packages/app/src/lib/modeled-system-power.test.ts +++ b/packages/app/src/lib/modeled-system-power.test.ts @@ -222,6 +222,89 @@ describe('modeled system power admission and accounting', () => { expect(modelSystemPower(source)).toMatchObject({ reason: 'role-power' }); }); + it('models an aggregate multinode deployment without worker telemetry at the deployment mean', () => { + // Kimi K3 B200 dynamo-vLLM TP8/PP2 (prod rows, 2026-09-17): two eight-GPU hosts, + // aggregate producer, no per-worker array. The K3 H200 vLLM row below is + // TP16 × 2 DP replicas across four hosts. + const b200 = row({ + is_multinode: true, + num_prefill_gpu: 16, + num_decode_gpu: 16, + decode_num_workers: 1, + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 715.095, + avg_total_gpu_power_w: 11441.513, + decode_pp: 2, + }, + }); + const perChassis = estimateChassisPower('b200', 11441.513 / 2, 1.3)!; + const estimate = modelSystemPower(b200); + expect(estimate).toMatchObject({ + status: 'supported', + topologyBasis: 'uniform-hosts', + chassisBasis: 'full', + gpuCount: 16, + chassisCount: 2, + modeledGpuCount: 16, + }); + if (estimate.status !== 'supported') throw new Error('unreachable'); + expect(estimate.chassisAcWatts).toBeCloseTo(perChassis.chassisAcWatts * 2, 6); + expect(estimate.deploymentFacilityWatts).toBe(estimate.facilityWatts); + expect(estimate.chassisAcWattsPerGpu).toBeCloseTo(perChassis.chassisAcWatts / 8, 6); + + const h200 = row({ + hardware: 'h200', + is_multinode: true, + prefill_tp: 16, + decode_tp: 16, + decode_ep: 32, + decode_dp_attention: true, + decode_num_workers: 2, + num_prefill_gpu: 32, + num_decode_gpu: 32, + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 167.357, + avg_total_gpu_power_w: 5355.413, + decode_pp: 1, + }, + }); + expect(modelSystemPower(h200)).toMatchObject({ + status: 'supported', + topologyBasis: 'uniform-hosts', + chassisCount: 4, + gpuCount: 32, + }); + + // A replica count that does not explain the telemetry width is not guessed around. + expect(modelSystemPower({ ...h200, decode_num_workers: 1 })).toMatchObject({ + reason: 'gpu-count', + }); + // Twelve GPUs cannot fill whole eight-GPU hosts; placement is unknown. + const twelve = row({ is_multinode: true, decode_tp: 12, prefill_tp: 12 }); + twelve.metrics.avg_total_gpu_power_w = twelve.metrics.avg_power_w * 12; + expect(modelSystemPower(twelve)).toMatchObject({ reason: 'topology' }); + // Per-worker telemetry, when present, keeps the more exact worker path. + const withWorkers = { + ...b200, + workers: ['host-a', 'host-b'].map((host, worker_idx) => ({ + role: 'agg', + worker_idx, + hosts: [host], + num_gpus: 8, + avg_power_w: 715.095, + })), + }; + expect(modelSystemPower(withWorkers)).toMatchObject({ + status: 'supported', + topologyBasis: 'worker-hosts', + chassisCount: 2, + }); + }); + it('preserves meaningful aggregate PP and PCP aliases before checking physical width', () => { for (const widths of [ { decode_pp: 1, prefill_pp: 2 }, diff --git a/packages/app/src/lib/modeled-system-power.ts b/packages/app/src/lib/modeled-system-power.ts index 550fe23fc..aadb737f6 100644 --- a/packages/app/src/lib/modeled-system-power.ts +++ b/packages/app/src/lib/modeled-system-power.ts @@ -43,7 +43,13 @@ export type SystemPowerEstimate = deploymentFacilityWatts: number; pue: number; telemetryBasis: 'validated-v2' | 'validated-unversioned-single-node'; - topologyBasis: 'single-node' | 'worker-hosts'; + /** + * 'single-node': one host, one chassis. 'worker-hosts': one chassis per + * measured worker, each at its own telemetry. 'uniform-hosts': an + * aggregate multinode deployment whose producer emitted no per-worker + * telemetry; every eight-GPU chassis is modeled at the deployment mean. + */ + topologyBasis: 'single-node' | 'worker-hosts' | 'uniform-hosts'; /** * 'full': every chassis had all eight GPUs measured. 'extrapolated': at least * one chassis was partially allocated; its model input is the measured per-GPU @@ -138,7 +144,7 @@ export function modelSystemPower( } const chassis: MeasuredChassis[] = []; - let topologyBasis: 'single-node' | 'worker-hosts'; + let topologyBasis: 'single-node' | 'worker-hosts' | 'uniform-hosts'; if (row.disagg === false && row.is_multinode === false) { // One host cannot hold more than one chassis. if (gpuCount > CHASSIS_GPU_COUNT) return unavailable('topology'); @@ -173,6 +179,36 @@ export function modelSystemPower( ? m.avg_total_gpu_power_w : m.avg_power_w * CHASSIS_GPU_COUNT, }); + } else if (row.disagg === false && (!Array.isArray(row.workers) || row.workers.length === 0)) { + // Aggregate multinode producers emit no per-worker telemetry. Symmetric + // TP/PP/DP shards load every host alike, so each full eight-GPU chassis is + // modeled at the deployment mean; the supported hardware only ships in + // eight-GPU hosts, so the count must fill whole chassis on several hosts. + // Disaggregated roles differ in load and stay on the worker path. + const hostCount = gpuCount / CHASSIS_GPU_COUNT; + if (!count(hostCount) || hostCount < 2) return unavailable('topology'); + const tp = row.decode_tp > 0 ? row.decode_tp : row.prefill_tp; + const pp = Math.max(m.pp ?? 1, m.decode_pp ?? 1, m.prefill_pp ?? 1); + const pcp = Math.max(m.pcp_size ?? 1, m.decode_pcp_size ?? 1, m.prefill_pcp_size ?? 1); + // Data-parallel replicas widen the deployment beyond one TP×PP×PCP group. + const replicas = Math.max(1, row.decode_num_workers); + if ( + !count(tp) || + !count(pp) || + !count(pcp) || + !count(replicas) || + tp * pp * pcp * replicas !== gpuCount + ) { + return unavailable('gpu-count'); + } + topologyBasis = 'uniform-hosts'; + for (let host = 0; host < hostCount; host++) { + chassis.push({ + measuredGpus: CHASSIS_GPU_COUNT, + // Partition the producer's exact total so the chassis inputs sum back to it. + modelInputWatts: m.avg_total_gpu_power_w / hostCount, + }); + } } else { // A role average across several hosts is insufficient for nonlinear // fan/PSU evaluation. Require one chassis per measured worker and a diff --git a/packages/app/src/lib/power-basis.test.ts b/packages/app/src/lib/power-basis.test.ts new file mode 100644 index 000000000..0aa67a4f5 --- /dev/null +++ b/packages/app/src/lib/power-basis.test.ts @@ -0,0 +1,73 @@ +import { describe, expect, it } from 'vitest'; + +import type { BenchmarkRow } from '@/lib/api'; +import { rowToAggDataEntry, transformBenchmarkRows } from '@/lib/benchmark-transform'; +import { buildDerivedChartFields, getHardwareKey } from '@/lib/chart-utils'; +import { POWER_BASIS_FIELDS } from '@/lib/power-basis'; + +// Qwen3.5 B200 c1, run 34175132645: actual rounded telemetry, eight GPUs +// (same fixture as modeled-system-power.test.ts) plus an output rate. +function row(overrides: Partial = {}): BenchmarkRow { + return { + id: 441192, + model: 'qwen3.5', + hardware: 'b200', + framework: 'sglang', + precision: 'fp8', + spec_method: 'none', + disagg: false, + is_multinode: false, + prefill_tp: 8, + prefill_ep: 1, + prefill_dp_attention: false, + prefill_num_workers: 0, + decode_tp: 8, + decode_ep: 1, + decode_dp_attention: false, + decode_num_workers: 0, + num_prefill_gpu: 8, + num_decode_gpu: 8, + benchmark_type: 'single_turn', + isl: 8192, + osl: 1024, + conc: 1, + offload_mode: 'off', + image: 'lmsysorg/sglang:v0.5.19-cu130', + date: '2026-09-08', + run_url: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34175132645/attempts/1', + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 349.859, + avg_total_gpu_power_w: 2798.868, + total_gpu_energy_j: 120361.299, + joules_per_output_token: 12.937902, + tput_per_gpu: 270, + output_tput_per_gpu: 30, + pp: 1, + pcp_size: 1, + median_intvty: 100, + }, + ...overrides, + }; +} + +/** Full official-path derivation for one row with real HW_REGISTRY specs. */ +function derive(source: BenchmarkRow) { + const entry = rowToAggDataEntry(source); + const hwKey = getHardwareKey(entry); + return { entry, hwKey, fields: buildDerivedChartFields(entry, hwKey) }; +} + +describe('power boundaries through the derived-field builder', () => { + it('serves the same fields to ?unofficialrun= overlays through transformBenchmarkRows', () => { + const { chartData } = transformBenchmarkRows([row()], 'median', 'external'); + const point = chartData[0][0]; + const official = derive(row()).fields; + for (const basis of Object.values(POWER_BASIS_FIELDS)) { + expect(point[basis.watts]).toEqual(official[basis.watts]); + expect(point[basis.energy]).toEqual(official[basis.energy]); + expect(point[basis.watts]?.roof).toBe(false); + } + }); +}); diff --git a/packages/app/src/lib/power-basis.ts b/packages/app/src/lib/power-basis.ts new file mode 100644 index 000000000..034eeca2d --- /dev/null +++ b/packages/app/src/lib/power-basis.ts @@ -0,0 +1,211 @@ +/** + * Power boundaries for one benchmark point, from the GPU board out to the + * utility meter. Pure numbers only — no React, DOM, or registry imports — so + * the derived-field builder, overlays, and tests share one formula set. + * + * | Basis | W / GPU | J / output token | + * | ---------------------- | ------------------------------------------------ | ----------------------------------------- | + * | B1 gpu-measured | `avg_power_w` (existing `measuredAvgPower`) | `joules_per_output_token` (existing) | + * | B2 gpu-provisioned | `HW_REGISTRY.tdp` | W × N_alloc ÷ total output tok/s | + * | B3 utility-provisioned | `HW_REGISTRY.power` × 1000 | W × N_alloc ÷ total output tok/s | + * | B4 utility-modeled | modeled `deploymentFacilityWatts` ÷ `gpuCount` | B1 J/out × (B4 W ÷ B1 W) | + * + * N_alloc counts every allocated GPU (prefill + decode for disaggregation); + * total output tok/s is the whole deployment's. B4 reuses the estimate that + * `modelSystemPower` already attached to the entry — PUE is applied exactly + * once inside that model, never here — and is withheld wherever B1 is. + * Unavailable values are `null`; callers omit the field rather than plotting 0. + */ +import type { AggDataEntry, InferenceData, PowerBasisFieldKey } from '@/components/inference/types'; + +export const POWER_BASES = [ + 'gpu-measured', + 'gpu-provisioned', + 'utility-provisioned', + 'utility-modeled', +] as const; +export type PowerBasis = (typeof POWER_BASES)[number]; +export type PowerQuantity = 'watts' | 'energy'; + +export const POWER_BASIS_LABELS: Record = { + 'gpu-measured': { en: 'GPU measured', zh: 'GPU 实测' }, + 'gpu-provisioned': { en: 'GPU provisioned (TDP)', zh: 'GPU 额定(TDP)' }, + 'utility-provisioned': { en: 'Utility provisioned (all-in)', zh: '全电源配置(all-in)' }, + 'utility-modeled': { en: 'Utility modeled (PUE)', zh: '数据中心建模(含 PUE)' }, +}; + +/** InferenceData keys per derived basis and quantity. B1 lives on the measured* fields. */ +export const POWER_BASIS_FIELDS: Record< + Exclude, + Record +> = { + 'gpu-provisioned': { + watts: 'gpuProvisionedWatts', + energy: 'gpuProvisionedJPerOutputToken', + }, + 'utility-provisioned': { + watts: 'utilityProvisionedWatts', + energy: 'utilityProvisionedJPerOutputToken', + }, + 'utility-modeled': { + watts: 'utilityModeledWatts', + energy: 'utilityModeledJPerOutputToken', + }, +}; + +export interface PowerBasisInput { + /** B2 W/GPU: HW_REGISTRY tdp. 0 means the spec is not yet available. */ + tdpWatts: number | null; + /** B3 W/GPU: HW_REGISTRY all-in power, already in watts. */ + utilityWatts: number | null; + /** Every GPU the deployment occupies (prefill + decode for disaggregation). */ + allocatedGpus: number | null; + /** Whole-deployment successful output tokens per second. */ + totalOutputTokPerSec: number | null; + /** B1 W/GPU from validated telemetry. */ + measuredWatts: number | null; + /** B1 J/output token from the same telemetry window. */ + measuredJPerOutputToken: number | null; + /** + * B4 W/GPU: modeled facility watts (PUE already applied) per measured GPU. + * B4 is a scaling of B1, so it is withheld whenever `measuredWatts` is null. + */ + modeledFacilityWattsPerGpu: number | null; +} + +export type PowerBasisValues = Record; + +const positive = (value: unknown): value is number => + typeof value === 'number' && Number.isFinite(value) && value > 0; +const count = (value: unknown): value is number => positive(value) && Number.isSafeInteger(value); +const orNull = (value: number): number | null => (positive(value) ? value : null); + +/** + * Derives the B2–B4 boundary values from plain numbers. Any unavailable input + * yields `null` for the values that depend on it and leaves the rest intact. + */ +export function computePowerBasisFields(input: PowerBasisInput): PowerBasisValues { + const tdp = positive(input.tdpWatts) ? input.tdpWatts : null; + const utility = positive(input.utilityWatts) ? input.utilityWatts : null; + const measuredWatts = positive(input.measuredWatts) ? input.measuredWatts : null; + const measuredJ = positive(input.measuredJPerOutputToken) ? input.measuredJPerOutputToken : null; + // B4 is B1 carried out to the utility meter, so it follows B1's availability: + // no measured watts, no modeled boundary (B3 ≥ B4 ≥ B1 needs its anchor). + const modeled = + measuredWatts !== null && positive(input.modeledFacilityWattsPerGpu) + ? input.modeledFacilityWattsPerGpu + : null; + + // Provisioned energy: GPU-seconds spent per output token by the whole + // deployment (N_alloc ÷ total tok/s) × W per GPU = J per output token. + const gpuSecondsPerOutputToken = + positive(input.allocatedGpus) && positive(input.totalOutputTokPerSec) + ? input.allocatedGpus / input.totalOutputTokPerSec + : null; + const provisionedEnergy = (watts: number | null) => + watts !== null && gpuSecondsPerOutputToken !== null + ? orNull(watts * gpuSecondsPerOutputToken) + : null; + + // Modeled energy scales the producer's same-window E/N by modeled ÷ measured W, + // so it inherits B1's token denominator instead of re-deriving one. + const modeledEnergy = + modeled !== null && measuredWatts !== null && measuredJ !== null + ? orNull((measuredJ * modeled) / measuredWatts) + : null; + + return { + gpuProvisionedWatts: tdp, + gpuProvisionedJPerOutputToken: provisionedEnergy(tdp), + utilityProvisionedWatts: utility, + utilityProvisionedJPerOutputToken: provisionedEnergy(utility), + utilityModeledWatts: modeled, + utilityModeledJPerOutputToken: modeledEnergy, + }; +} + +type PowerBasisEntry = Pick< + AggDataEntry, + | 'output_tput_per_gpu' + | 'disagg' + | 'benchmark_type' + | 'num_prefill_gpu' + | 'num_decode_gpu' + | 'avg_power_w' + | 'joules_per_output_token' + | 'modeledSystemPower' +>; + +/** + * Whole-deployment normalization for the provisioned energies. Aggregate rows + * already report output per allocated GPU, so N_alloc cancels and the ratio + * 1 GPU : per-GPU throughput is exact without trusting display counts (legacy + * ingest can encode TP × EP twice). Fixed-sequence disaggregated rows report + * output per decode GPU while the deployment also powers the prefill pool, so + * total output = per-GPU × decode GPUs and N_alloc = prefill + decode GPUs. + * Other disaggregated benchmark types are left out: whether AgentX throughput + * already divides by all GPUs is not verifiable in-app. + */ +export function powerBasisNormalization( + entry: Pick< + PowerBasisEntry, + 'output_tput_per_gpu' | 'disagg' | 'benchmark_type' | 'num_prefill_gpu' | 'num_decode_gpu' + >, +): Pick { + const perGpu = entry.output_tput_per_gpu; + const unavailable = { allocatedGpus: null, totalOutputTokPerSec: null }; + if (!positive(perGpu)) return unavailable; + if (!entry.disagg) return { allocatedGpus: 1, totalOutputTokPerSec: perGpu }; + if (entry.benchmark_type !== 'single_turn') return unavailable; + const prefill = entry.num_prefill_gpu; + const decode = entry.num_decode_gpu; + if (!count(prefill) || !count(decode)) return unavailable; + return { allocatedGpus: prefill + decode, totalOutputTokPerSec: perGpu * decode }; +} + +/** + * B4 W/GPU from the estimate `rowToAggDataEntry` attached. The model owns + * telemetry admission: `modelSystemPower` requires `power_valid === 1` plus + * schema v2, or the validated unversioned single-node producer it records as + * `telemetryBasis: 'validated-unversioned-single-node'`. That is the same + * population the app plots as B1 (`measuredAvgPower`) and as + * `modeledChassisPowerPerGpu`, so B4 renders exactly where they do. The public + * API's stricter `strictV2` row filter is not re-applied here; it is not + * applied to the chart's B1 either. + */ +export function modeledFacilityWattsPerGpu( + entry: Pick, +): number | null { + const model = entry.modeledSystemPower; + if (model?.status !== 'supported') return null; + if (!positive(model.deploymentFacilityWatts) || !count(model.gpuCount)) return null; + return orNull(model.deploymentFacilityWatts / model.gpuCount); +} + +export type PowerBasisChartFields = Partial>; + +/** + * Chart-shaped B2–B4 fields for one entry. Keys are present only for finite, + * positive values: the metric filters drop a point by `metricKey in point`, + * and the coordinate remap falls back to raw throughput when a key exists + * with an unusable value. + */ +export function buildPowerBasisChartFields( + entry: PowerBasisEntry, + specs: { tdp?: number; power?: number }, +): PowerBasisChartFields { + const values = computePowerBasisFields({ + tdpWatts: specs.tdp ?? null, + utilityWatts: positive(specs.power) ? specs.power * 1000 : null, + ...powerBasisNormalization(entry), + measuredWatts: entry.avg_power_w ?? null, + measuredJPerOutputToken: entry.joules_per_output_token ?? null, + modeledFacilityWattsPerGpu: modeledFacilityWattsPerGpu(entry), + }); + const fields: PowerBasisChartFields = {}; + for (const key of Object.keys(values) as PowerBasisFieldKey[]) { + const y = values[key]; + if (y !== null) fields[key] = { y, roof: false }; + } + return fields; +} diff --git a/packages/app/src/lib/url-state.ts b/packages/app/src/lib/url-state.ts index da7bb00d3..78755aa95 100644 --- a/packages/app/src/lib/url-state.ts +++ b/packages/app/src/lib/url-state.ts @@ -63,6 +63,12 @@ const URL_STATE_KEYS = [ 'i_spec', // Measured-power certification tiers ('certified' / 'legacy', comma-joined). 'i_power', + // Completed Perf Rulers on the primary inference chart: `isoX|curveA|curveB` + // entries joined by `;` (see serializePerfRulers in d3-chart/layers/perf-ruler). + 'i_rulers', + // Comparison series overlaid on a gated power metric: `boundaries` (every + // power boundary) or `roles` (prefill / decode pools). Empty = the metric alone. + 'i_pcompare', // Exact serving-envelope pair behind an Overview 30-day comparison cell. 'i_overview_current', 'i_overview_baseline', @@ -177,6 +183,8 @@ export const PARAM_DEFAULTS: Record = { i_disagg: '', i_spec: '', i_power: '', + i_rulers: '', + i_pcompare: '', i_overview_current: '', i_overview_baseline: '', e_rundate: '', @@ -492,6 +500,26 @@ export function rememberChartStateInUrl(): string { return chartParams.toString(); } +/** + * The current page's URL carrying its chart state plus `overrides`, + * canonicalised like `rememberChartStateInUrl`: chart params and both + * unofficial-run spellings are dropped from the live address bar before the + * store's state (and the overrides) are layered on. For anchors that must + * work with open-in-new-tab, where the in-memory state would otherwise be lost. + */ +export function chartStateHref(overrides: Record): string { + const { origin, pathname, hash, search } = window.location; + const merged = new URLSearchParams(search); + for (const key of URL_STATE_KEYS) merged.delete(key); + // Collected first: deleting while iterating the params would skip entries. + const staleRunKeys = [...merged.keys()].filter((key) => UNOFFICIAL_RUN_PARAM_RE.test(key)); + for (const key of staleRunKeys) merged.delete(key); + for (const [key, value] of collectTabParams()) merged.set(key, value); + for (const [key, value] of Object.entries(overrides)) merged.set(key, value); + const query = merged.toString(); + return `${origin}${pathname}${query ? `?${query}` : ''}${hash}`; +} + /** * Append the current chart state to an outbound in-app href, so the page it * opens can link back to the chart the user left. Used for the agentic