From 1ba550d498b7aa389af2f32230e6ea7a252afc87 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Tue, 29 Sep 2026 12:27:46 -0700 Subject: [PATCH 1/7] feat(ui): power boundary metrics, comparison overlays and role energy MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add gated power-boundary y-axis metrics (GPU-measured, GPU-provisioned, utility provisioned, utility modeled), the i_pcompare boundary/role comparison overlays with per-series line labels, reconstructed prefill/decode energy, and CSV/table plumbing; overlay runs follow the same paths. 中文:新增门控的功耗边界 y 轴指标(GPU 实测、GPU 配给、市电配给、市电建模)、i_pcompare 边界/角色对比叠加与逐序列线标签、重建的 prefill/decode 能耗及 CSV/表格贯通;非正式 run 叠加走同一路径。 --- docs/index.md | 1 + .../component/inference-chart-controls.cy.tsx | 6 +- .../cypress/component/power-compare.cy.tsx | 154 +++++ .../cypress/component/scatter-graph.cy.tsx | 19 +- packages/app/cypress/support/mock-data.ts | 2 + .../src/components/calculator/profit-power.ts | 8 +- .../components/inference/InferenceContext.tsx | 42 +- .../inference/axis-metric-explanations.ts | 110 +++ .../inference/hooks/useChartData.ts | 13 +- .../inference/measured-metric-config.test.ts | 37 +- .../inference/measured-metric-config.ts | 87 ++- .../inference/metric-registry.test.ts | 11 +- .../components/inference/metric-registry.ts | 105 ++- .../app/src/components/inference/types.ts | 63 ++ .../components/inference/ui/ChartControls.tsx | 11 +- .../components/inference/ui/ChartDisplay.tsx | 84 ++- .../src/components/inference/ui/GPUGraph.tsx | 62 +- .../inference/ui/InferenceTable.test.ts | 55 +- .../inference/ui/InferenceTable.tsx | 28 +- .../inference/ui/MeasuredMetricControls.tsx | 138 +++- .../inference/ui/PowerMetricAvailability.tsx | 161 ++++- .../ui/ScatterGraph.test-harness.tsx | 19 + .../components/inference/ui/ScatterGraph.tsx | 635 +++++++++++++++--- .../inference/ui/inference-table-sort.ts | 4 +- .../inference/ui/line-label-layer.test.ts | 7 +- .../inference/ui/line-label-layer.ts | 113 +++- .../ui/line-label-visibility.test.ts | 20 + .../inference/ui/line-label-visibility.ts | 14 +- .../app/src/components/inference/utils.ts | 22 +- .../inference/utils/best-series-per-sku.ts | 4 +- .../inference/utils/point-identity.ts | 38 ++ .../inference/utils/power-compare.ts | 326 +++++++++ .../inference/utils/powerCurves.test.ts | 8 +- .../components/inference/utils/powerCurves.ts | 40 +- .../inference/utils/role-energy.test.ts | 26 + .../components/inference/utils/role-energy.ts | 64 ++ .../utils/tooltip-utils.power-trace.test.ts | 94 +++ .../inference/utils/tooltipUtils.ts | 194 +++++- packages/app/src/lib/chart-utils.test.ts | 138 ++-- packages/app/src/lib/chart-utils.ts | 41 +- .../app/src/lib/csv-export-helpers.test.ts | 37 + packages/app/src/lib/csv-export-helpers.ts | 14 +- packages/app/src/lib/inference-labels.ts | 42 ++ .../app/src/lib/modeled-system-power.test.ts | 83 +++ packages/app/src/lib/modeled-system-power.ts | 40 +- packages/app/src/lib/power-basis.test.ts | 73 ++ packages/app/src/lib/power-basis.ts | 211 ++++++ packages/app/src/lib/url-state.ts | 28 + 48 files changed, 3249 insertions(+), 283 deletions(-) create mode 100644 packages/app/cypress/component/power-compare.cy.tsx create mode 100644 packages/app/src/components/inference/utils/power-compare.ts create mode 100644 packages/app/src/components/inference/utils/role-energy.test.ts create mode 100644 packages/app/src/components/inference/utils/role-energy.ts create mode 100644 packages/app/src/components/inference/utils/tooltip-utils.power-trace.test.ts create mode 100644 packages/app/src/lib/power-basis.test.ts create mode 100644 packages/app/src/lib/power-basis.ts diff --git a/docs/index.md b/docs/index.md index c67e877ae..2e6c0224a 100644 --- a/docs/index.md +++ b/docs/index.md @@ -9,6 +9,7 @@ Design rationale and non-obvious conventions. See [CLAUDE.md](../CLAUDE.md) for - [API Skill Examples](./inferencex-api-examples.md) — Install the public skill, query benchmarks, export measured PowerX data, and explain empty results - [PowerX System Power](./powerx-system-power.md) — Pinned chassis model, measured-input guards, assumptions, and reproducible article exports +- [PowerX Permanent View](./powerx-permanent-view.md) — Power boundaries as gated Measured Energy metrics, `i_metric`/`i_rulers` share links, missing-value states - [PowerX Persistence and Recovery](./powerx-persistence-recovery.md) — Telemetry receipts, migration prerequisites, and targeted repair - [API Skill Releases](./inferencex-skills-release.md) — Prepare an immutable package, verify clean installations and agent exports, and publish through the package-specific workflow - [API Skill Discovery](./inferencex-skills-discovery.md) — Accept or reject implicit skill discovery in fresh Codex and Claude Code projects diff --git a/packages/app/cypress/component/inference-chart-controls.cy.tsx b/packages/app/cypress/component/inference-chart-controls.cy.tsx index ded6bd9ba..4445d303b 100644 --- a/packages/app/cypress/component/inference-chart-controls.cy.tsx +++ b/packages/app/cypress/component/inference-chart-controls.cy.tsx @@ -469,8 +469,10 @@ describe('Inference ChartControls grouped measured metrics', () => { cy.get(`input[aria-label="${searchLabel}"]`).type(group); cy.get('[data-select-option][data-value^="y_measured"]').should(($options) => { const values = [...$options].map((option) => option.dataset.value); - expect(values).to.have.length(13); - expect(new Set(values).size).to.equal(13); + // Thirteen measured axes plus the Timeline display of measured power. + expect(values).to.have.length(14); + expect(new Set(values).size).to.equal(14); + expect(values).to.include('y_measuredPowerTimeline'); }); cy.get(`input[aria-label="${searchLabel}"]`).clear().type(power); cy.get('[data-select-option]') diff --git a/packages/app/cypress/component/power-compare.cy.tsx b/packages/app/cypress/component/power-compare.cy.tsx new file mode 100644 index 000000000..e1d3917a0 --- /dev/null +++ b/packages/app/cypress/component/power-compare.cy.tsx @@ -0,0 +1,154 @@ +import { PathnameContext } from 'next/dist/shared/lib/hooks-client-context.shared-runtime'; + +import ScatterGraph from '@/components/inference/ui/ScatterGraph'; +import type { InferenceData } from '@/components/inference/types'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; +import { Precision } from '@/lib/data-mappings'; +import { overlayRunColor } from '@/lib/overlay-run-style'; + +import { + createMockChartDefinition, + createMockHardwareConfig, + createMockInferenceData, +} from '../support/mock-data'; +import { mountWithProviders } from '../support/test-utils'; + +// Power comparison series (`i_pcompare`) on `?unofficialrun=` overlays: role +// clones draw in the overlay run colour with the role dash, and their legend +// rows toggle them. + +const OVERLAY_RUN_ID = 31415926535; +const OVERLAY_RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${OVERLAY_RUN_ID}`; +const hwConfig = createMockHardwareConfig(); +const chartDefinition = createMockChartDefinition({ + chartType: 'interactivity', + y_measuredAvgPower: 'measuredAvgPower.y', + y_measuredAvgPower_roofline: 'lower_left', +}); + +const metric = (y: number) => ({ y, roof: false }); + +/** + * Three measured points with both roles available. Power falls as x rises so + * every point sits on the upper power envelope the chart draws for watt axes. + */ +function measuredCurve(hwKey: string, run_url?: string): InferenceData[] { + return [ + [8, 700, 840, 500], + [16, 600, 820, 450], + [32, 500, 800, 400], + ].map(([x, measured, prefill, decode]) => + createMockInferenceData({ + hwKey, + x, + conc: x, + y: measured, + precision: Precision.FP4, + run_url, + disagg: true, + measuredAvgPower: metric(measured), + measuredPrefillAvgPower: metric(prefill), + measuredDecodeAvgPower: metric(decode), + }), + ); +} + +function mountCompare(data: InferenceData[], overlay: InferenceData[]) { + mountWithProviders( + +
+ +
+
, + { + inference: { + selectedYAxisMetric: 'y_measuredAvgPower', + hardwareConfig: hwConfig, + activeHwTypes: new Set(['b200', 'h100']), + hwTypesWithData: new Set(['b200', 'h100']), + selectedPrecisions: [Precision.FP4], + hideNonOptimal: false, + showLineLabels: false, + }, + unofficial: { + isUnofficialRun: true, + activeOverlayHwTypes: new Set(['h100']), + allOverlayHwTypes: new Set(['h100']), + runIndexByUrl: { [OVERLAY_RUN_URL]: 0, [String(OVERLAY_RUN_ID)]: 0 }, + unofficialRunInfos: [ + { + id: OVERLAY_RUN_ID, + name: 'powerx-compare', + branch: 'powerx-compare', + sha: 'abc000', + createdAt: '2026-09-01T00:00:00Z', + url: OVERLAY_RUN_URL, + conclusion: 'success', + status: 'completed', + isNonMainBranch: true, + }, + ], + }, + }, + ); +} + +const svg = '#power-compare-test svg'; +const legend = '#power-compare-test [data-testid="chart-legend"]'; + +describe('ScatterGraph power comparison series', () => { + beforeEach(() => { + cy.on('uncaught:exception', (error) => { + if (error.message.includes('ResizeObserver loop')) return false; + }); + }); + + it('draws role siblings for ?unofficialrun= overlays in the run colour with the role dash', () => { + const official = expandPowerCompareSeries(measuredCurve('b200'), 'y_measuredAvgPower', 'roles'); + const overlay = expandPowerCompareSeries( + measuredCurve('h100', OVERLAY_RUN_URL), + 'y_measuredAvgPower', + 'roles', + ); + mountCompare(official, overlay); + + cy.get(`${svg} .roofline-path[data-power-variant="prefill"]`).should( + 'have.attr', + 'stroke-dasharray', + '7 3', + ); + cy.get(`${svg} .unofficial-overlay-pt`).should('have.length', 9); + cy.get(`${svg} .overlay-roofline-path[data-power-variant="decode"]`) + .should('have.length', 1) + .and('have.attr', 'stroke', overlayRunColor(0)) + .and('have.attr', 'stroke-dasharray', '2 3'); + cy.get(`${svg} .overlay-roofline-path:not([data-power-variant])`).should('have.length', 1); + cy.get(legend).within(() => { + cy.contains('All GPUs').should('exist'); + cy.contains('Prefill GPUs').should('exist'); + cy.contains('Decode GPUs').click(); + }); + cy.get(`${svg} .overlay-roofline-path[data-power-variant="decode"]`).should( + 'have.css', + 'opacity', + '0', + ); + cy.get(`${svg} .unofficial-overlay-pt`).then(($points) => { + const hidden = [...$points].filter((point) => getComputedStyle(point).opacity === '0'); + expect(hidden).to.have.length(3); + }); + }); +}); diff --git a/packages/app/cypress/component/scatter-graph.cy.tsx b/packages/app/cypress/component/scatter-graph.cy.tsx index 0f157fea1..fac02fad5 100644 --- a/packages/app/cypress/component/scatter-graph.cy.tsx +++ b/packages/app/cypress/component/scatter-graph.cy.tsx @@ -925,14 +925,19 @@ describe('ScatterGraph', () => { cy.get('#test-scatter-overlay-labels svg .line-label') .filter('[data-line-key]:not([data-line-key^="overlay-"])') .should('have.length.greaterThan', 0); - // The exact branch that crashed the production page remains visible in the - // overlay line label and legend after ScatterGraph's render-time updates. - cy.get('#test-scatter-overlay-labels svg .line-label[data-line-key^="overlay-"]') - .find('text') - .should('contain.text', runBranch); + // The pill names the hardware behind the ✕ marker, parsed like an official + // pill; the long branch that crashed the production page stays in the legend. + cy.get('#test-scatter-overlay-labels svg .line-label[data-line-key^="overlay-"] .ll-text') + .should('have.text', '✕ B200 (TRTLLM)') + .and('not.contain.text', runBranch); cy.get( '#test-scatter-overlay-labels svg .line-label[data-line-key^="overlay-"] .ll-gpu', - ).should('not.exist'); + ).should('have.text', 'B200'); + // b200_trt is active only in the overlay legend (official rows: h100), so + // the overlay pill must stay visible after the filter-sync effect. + cy.get('#test-scatter-overlay-labels svg .line-label[data-line-key^="overlay-"]') + .should('have.attr', 'data-visible', '1') + .and('have.css', 'opacity', '1'); cy.get('#test-scatter-overlay-labels [data-testid="chart-legend"]').should( 'contain.text', runBranch, @@ -1094,7 +1099,7 @@ describe('ScatterGraph', () => { cy.get('#test-scatter-singleton-overlay-label svg .line-label[data-line-key^="overlay-"]') .should('have.length', 1) .find('text') - .should('contain.text', 'tileRT'); + .should('have.text', '✕ B200 (TRTLLM)'); cy.get('#test-scatter-singleton-overlay-label svg').then(($svg) => { const svg = $svg[0]; diff --git a/packages/app/cypress/support/mock-data.ts b/packages/app/cypress/support/mock-data.ts index 413f38fab..733d04e49 100644 --- a/packages/app/cypress/support/mock-data.ts +++ b/packages/app/cypress/support/mock-data.ts @@ -232,6 +232,8 @@ export function createMockInferenceContextValues( setSelectedXAxisMode: namedStub('setSelectedXAxisMode'), scaleType: 'auto', setScaleType: namedStub('setScaleType'), + powerCompare: 'none' as const, + setPowerCompare: namedStub('setPowerCompare'), quickFilters: { vendors: [], frameworks: [], deployment: [], spec: [], power: [] }, availableQuickFilters: { vendors: [], frameworks: [], deployment: [], spec: [], power: [] }, setQuickFilterVendors: namedStub('setQuickFilterVendors'), diff --git a/packages/app/src/components/calculator/profit-power.ts b/packages/app/src/components/calculator/profit-power.ts index 089abb453..1c6bc2a1b 100644 --- a/packages/app/src/components/calculator/profit-power.ts +++ b/packages/app/src/components/calculator/profit-power.ts @@ -40,8 +40,12 @@ function planningPower(point: GPUDataPoint): PlanningPower { } } } - // Full-chassis planning requires whole replicas to fit on one eight-GPU host. - if (estimate.topologyBasis !== 'single-node' || 8 % estimate.gpuCount !== 0) + // Partial allocations must tile one host; fully measured multi-host estimates + // retain their validated worker-hosts or uniform-hosts topology. + if ( + estimate.chassisBasis === 'extrapolated' && + (estimate.topologyBasis !== 'single-node' || 8 % estimate.gpuCount !== 0) + ) return { reason: 'unsupported-power-topology' }; return { kwPerGpu: (estimate.deploymentFacilityWatts / estimate.gpuCount / 1000) * 1.1, diff --git a/packages/app/src/components/inference/InferenceContext.tsx b/packages/app/src/components/inference/InferenceContext.tsx index 036f22700..206c9fd47 100644 --- a/packages/app/src/components/inference/InferenceContext.tsx +++ b/packages/app/src/components/inference/InferenceContext.tsx @@ -37,6 +37,7 @@ import type { InferenceDataContextType, InferenceDisplayContextType, InferenceFiltersContextType, + PowerCompare, TokenRevenuePriceSource, } from '@/components/inference/types'; import { resolveMetricConfigKey } from '@/components/inference/metric-registry'; @@ -56,6 +57,14 @@ import { useUrlStateSync, } from '@/hooks/useChartContext'; import { useUrlState } from '@/hooks/useUrlState'; +import { serializePerfRulers } from '@/lib/d3-chart/layers/perf-ruler'; +import { parsePowerCompare } from '@/components/inference/utils/power-compare'; +import { + PERSISTED_PERF_RULER_CHART_ID, + PerfRulerStoreContext, + persistedPerfRulerAxisKey, + usePerfRulerStoreValue, +} from '@/components/inference/perf-ruler-store'; import { useParetoHighlightToggle } from './hooks/useParetoHighlightToggle'; import { useOpenRouterPricing } from '@/hooks/api/use-openrouter-pricing'; import { DEFAULT_Y_AXIS_METRIC } from '@/lib/url-state'; @@ -483,6 +492,12 @@ export function InferenceProvider({ const [scaleType, setScaleType] = useState<'auto' | 'linear' | 'log'>( () => (getUrlParam('i_scale') as 'auto' | 'linear' | 'log') || 'auto', ); + // Comparison series on a gated power metric (`i_pcompare`). Kept while the + // metric changes: a key without a common axis simply yields no siblings, and + // the Measured controls say so, so a link's intent survives a detour. + const [powerCompare, setPowerCompare] = useState(() => + parsePowerCompare(getUrlParam('i_pcompare')), + ); // ── Quick filters (vendor / framework / deployment / mtp-stp / power tier) ── // Coarse pre-filters applied to the point set. Empty = no constraint. @@ -770,6 +785,7 @@ export function InferenceProvider({ !isUnofficialRun && !hasExplicitRunSelection && selectedRunDateRev === 0, + powerCompare, ); // For GPU comparison date picker — use shared availability data from global filters @@ -1035,6 +1051,21 @@ export function InferenceProvider({ const refreshing = !availabilityError && chartDataRefreshing; const error = availabilityError || workflowError || chartDataError; + // ── Perf rulers (persisted chart) ──────────────────────────────────────── + // The axis identity follows the graph ChartDisplay renders as `chart-0` + // (picked by x mode, like `bestHwTypes` below), so an x-mode switch that + // swaps the rendered chart or its x units clears the rulers the same way + // the chart's own `usePerfRulerAxisReset` does for local state. + const perfRulerStore = usePerfRulerStoreValue( + PERSISTED_PERF_RULER_CHART_ID, + getUrlParam('i_rulers'), + persistedPerfRulerAxisKey(graphs, selectedXAxisMode, selectedYAxisMetric), + ); + const iRulersStr = useMemo( + () => serializePerfRulers(perfRulerStore.state), + [perfRulerStore.state], + ); + // ── Toggle sets ─────────────────────────────────────────────────────────── const { @@ -1603,6 +1634,8 @@ export function InferenceProvider({ i_disagg: quickFilterDeployment.join(','), i_spec: quickFilterSpec.join(','), i_power: quickFilterPower.join(','), + i_rulers: iRulersStr, + i_pcompare: powerCompare === 'none' ? '' : powerCompare, }, [ selectedYAxisMetric, @@ -1634,6 +1667,8 @@ export function InferenceProvider({ quickFilterDeployment, quickFilterSpec, quickFilterPower, + iRulersStr, + powerCompare, ], ); @@ -1849,6 +1884,7 @@ export function InferenceProvider({ selectedE2eXAxisMetric, selectedXAxisMode, scaleType, + powerCompare, isLegendExpanded, hideNonOptimal, showAllMeasurements, @@ -1874,6 +1910,7 @@ export function InferenceProvider({ selectedE2eXAxisMetric, selectedXAxisMode, scaleType, + powerCompare, isLegendExpanded, hideNonOptimal, showAllMeasurements, @@ -1908,6 +1945,7 @@ export function InferenceProvider({ setSelectedXAxisMetric, setSelectedXAxisMode: handleSetXAxisMode, setScaleType, + setPowerCompare, setQuickFilterVendors, setQuickFilterFrameworks, setQuickFilterDeployment, @@ -1944,7 +1982,9 @@ export function InferenceProvider({ display={displayValue} actions={actionsValue} > - {children} + + {children} + = { zh: '% TDP = 每芯片实测平均功耗(W)÷ 额定 TDP(W)× 100', }, }, + measuredPowerTimeline: { + description: { + en: + `The per-second accelerator power samples behind each measured average, drawn over ` + + `the whole benchmark job (server start, warmup, and the validated measurement window, ` + + `which is emphasized). One trace per config, mean of its GPUs by default; the rated TDP ` + + `is a dashed reference per hardware. Configs whose telemetry artifact is missing are ` + + `listed under the chart rather than estimated.${MEASURED_TIER_NOTE_EN}`, + zh: + `每个实测平均值背后的逐秒加速器功耗采样,覆盖整个基准测试任务(服务启动、warmup ` + + `以及被突出显示的有效测量窗口)。每个配置一条曲线,默认取其 GPU 的平均值;` + + `每种硬件的额定 TDP 以虚线作为参考。缺少遥测产物的配置会列在图表下方,而不会用估算值代替。${ + MEASURED_TIER_NOTE_ZH + }`, + }, + formula: { + en: 'W(t) = mean over GPUs of the sampled power draw in each one-second bucket', + zh: 'W(t) = 每个一秒时间桶内各 GPU 功耗采样值的平均', + }, + }, + gpuProvisionedWatts: { + description: { + en: + 'Rated accelerator TDP from the hardware registry, shown as a flat per-chip value so ' + + 'measured power can be read against the GPU-only provisioning boundary. It does not ' + + 'depend on the run.', + zh: + '取硬件注册表中的加速器额定 TDP,以每芯片恒定值显示,用于对照 GPU 侧的额定供电边界与实测功耗。' + + '该值与具体运行无关。', + }, + formula: { + en: 'W/GPU = rated TDP (W)', + zh: 'W/GPU = 额定 TDP(W)', + }, + }, + gpuProvisionedJPerOutputToken: { + description: { + en: + 'Energy per output token if every allocated accelerator drew exactly its rated TDP for ' + + 'the whole run. Disaggregated deployments count prefill and decode GPUs together, so ' + + 'this is the GPU-only provisioning boundary the measured J/token can be compared against.', + zh: + '假设所有已分配加速器在整个运行中恒以额定 TDP 耗电时的每输出 token 能耗。' + + '分离式部署将 prefill 与 decode GPU 一并计入,因此它是可与实测 J/token 对照的 GPU 侧额定边界。', + }, + formula: { + en: 'J/tok = rated TDP (W) × allocated GPUs ÷ total output tokens per second', + zh: 'J/tok = 额定 TDP(W)× 已分配 GPU 数 ÷ 总输出 token 吞吐(tok/s)', + }, + }, + utilityProvisionedWatts: { + description: { + en: + 'All-in provisioned power per chip from the hardware registry: the utility-side capacity ' + + 'a data center reserves for one accelerator including host, networking, cooling and ' + + 'power-conversion overheads. It is a flat value independent of the run.', + zh: + '取硬件注册表中的每芯片全电源配置功耗:数据中心为单张加速器预留的电源侧容量,' + + '包含主机、网络、散热与电源转换开销。该值为恒定值,与运行无关。', + }, + formula: { + en: 'W/GPU = all-in provisioned power per GPU (kW) × 1000', + zh: 'W/GPU = 每 GPU 全电源配置功耗(kW)× 1000', + }, + }, + utilityProvisionedJPerOutputToken: { + description: { + en: + 'Energy per output token at the all-in provisioned power boundary, normalized by every ' + + 'allocated accelerator. It differs from the public All-in Provisioned J per Output Token ' + + 'metric only for disaggregated runs, where that metric normalizes by decode GPUs alone.', + zh: + '在全电源配置边界下的每输出 token 能耗,按全部已分配加速器归一。' + + '仅在分离式运行中与公开的 All-in Provisioned J per Output Token 指标不同,后者只按 decode GPU 归一。', + }, + formula: { + en: 'J/tok = all-in provisioned power per GPU (W) × allocated GPUs ÷ total output tokens per second', + zh: 'J/tok = 每 GPU 全电源配置功耗(W)× 已分配 GPU 数 ÷ 总输出 token 吞吐(tok/s)', + }, + }, + utilityModeledWatts: { + description: { + en: + 'Modeled facility power per allocated accelerator: measured GPU power is scaled to chassis ' + + 'AC by the system power model and then multiplied once by PUE. Only hardware with a known ' + + 'eight-GPU chassis profile on 8k1k runs is supported; NVL72 systems show no value.', + zh: + '每已分配加速器的数据中心建模功耗:先由系统功耗模型将 GPU 实测功耗换算为机箱交流功耗,再乘以一次 PUE。' + + '仅支持在 8k1k 运行中具有已知八卡机箱模型的硬件;NVL72 系统不显示数值。', + }, + formula: { + en: 'W/GPU = modeled chassis AC power (W) × PUE ÷ allocated GPUs', + zh: 'W/GPU = 机箱交流建模功耗(W)× PUE ÷ 已分配 GPU 数', + }, + }, + utilityModeledJPerOutputToken: { + description: { + en: + 'Measured energy per output token scaled to the modeled facility boundary, so its ratio to ' + + 'measured GPU energy equals the ratio of modeled facility power to measured GPU power. ' + + 'Missing where the system power model or validated measured power is unavailable.', + zh: + '将实测每输出 token 能耗按建模的数据中心边界缩放,其与 GPU 实测能耗之比等于数据中心建模功耗与 GPU 实测功耗之比。' + + '系统功耗模型或通过验证的实测功耗缺失时不显示。', + }, + formula: { + en: 'J/tok = measured J per output token × modeled facility W per GPU ÷ measured W per GPU', + zh: 'J/tok = 实测每输出 token 能耗 × 每 GPU 数据中心建模功耗(W)÷ 每 GPU 实测功耗(W)', + }, + }, }; /** diff --git a/packages/app/src/components/inference/hooks/useChartData.ts b/packages/app/src/components/inference/hooks/useChartData.ts index 9f6498b5a..791b14be6 100644 --- a/packages/app/src/components/inference/hooks/useChartData.ts +++ b/packages/app/src/components/inference/hooks/useChartData.ts @@ -27,12 +27,14 @@ import type { ChartDefinition, HardwareConfig, InferenceData, + PowerCompare, RenderableGraph, TokenRevenuePriceSource, TokenRevenuePricing, YAxisMetricKey, } from '@/components/inference/types'; import { partitionChartDataByLimits } from '@/components/inference/utils'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; import { parseComparisonEntry } from '@/components/inference/utils/comparisonEntry'; import { computeAvailableQuickFilters, @@ -116,6 +118,8 @@ export function useChartData( tcoBasis: TcoBasis = DEFAULT_TCO_BASIS, /** Opt-in from the inference page only; explicit date/run/history views opt out. */ allowDefaultRunPreference = false, + /** Sibling boundary / role series appended to a gated power metric (`i_pcompare`). */ + powerCompare: PowerCompare = 'none', ) { // When the selected date is the latest available, use '' (empty string) to match // the initial no-date query key, reusing the eagerly-fetched benchmarks from the @@ -538,8 +542,14 @@ export function useChartData( ); const hasMetric = metricData.length > 0; const isTtftX = typeof xAxisField === 'string' && xAxisField.endsWith('_ttft'); + // Comparison clones are appended after the remap so they share the + // base point's x and differ only in y and `powerVariant`. const mappedData = hasMetric - ? metricData.map((d) => remapInferencePoint(d, metricKey, xAxisField)) + ? expandPowerCompareSeries( + metricData.map((d) => remapInferencePoint(d, metricKey, xAxisField)), + selectedYAxisMetric, + powerCompare, + ) : []; const isAgentic = selectedSequence === Sequence.AgenticTraces; @@ -576,6 +586,7 @@ export function useChartData( compareGpuPair, selectedPercentile, quickFilters, + powerCompare, ]); // Points that pass every scope filter but NOT the y-metric coverage filter. diff --git a/packages/app/src/components/inference/measured-metric-config.test.ts b/packages/app/src/components/inference/measured-metric-config.test.ts index 47c356f22..41276d400 100644 --- a/packages/app/src/components/inference/measured-metric-config.test.ts +++ b/packages/app/src/components/inference/measured-metric-config.test.ts @@ -1,6 +1,10 @@ import { describe, expect, it } from 'vitest'; -import { MEASURED_ENERGY_METRIC_CONFIG_KEYS, METRIC_CONFIG_KEYS } from './metric-registry'; +import { + MEASURED_ENERGY_METRIC_CONFIG_KEYS, + METRIC_CONFIG_KEYS, + POWER_BASIS_METRIC_CONFIG_KEYS, +} from './metric-registry'; import { changeMeasuredMetricConfig, getMeasuredMetricConfig, @@ -8,20 +12,27 @@ import { } from './measured-metric-config'; describe('measured metric configuration', () => { - it.each(MEASURED_ENERGY_METRIC_CONFIG_KEYS)( - 'round-trips the existing share-link metric %s', - (key) => { - const config = getMeasuredMetricConfig(key); - expect(config).toBeDefined(); - expect(changeMeasuredMetricConfig(key, {})).toBe(key); - expect(changeMeasuredMetricConfig('y_tpPerGpu', config!)).toBe(key); - }, - ); + it.each([ + 'y_measuredAvgPower', + 'y_measuredPowerTimeline', + 'y_measuredJPerOutputToken', + 'y_measuredWhPerSuccessfulQuery', + ] as const)('round-trips the existing share-link metric %s', (key) => { + const config = getMeasuredMetricConfig(key); + expect(config).toBeDefined(); + expect(changeMeasuredMetricConfig(key, {})).toBe(key); + expect(changeMeasuredMetricConfig('y_tpPerGpu', config!)).toBe(key); + }); it('does not group unrelated metrics or unknown persisted values', () => { const grouped = METRIC_CONFIG_KEYS.filter((key) => getMeasuredMetricConfig(key)); - expect(grouped).toHaveLength(13); - expect(new Set(grouped)).toEqual(new Set(MEASURED_ENERGY_METRIC_CONFIG_KEYS)); + expect(grouped).toHaveLength(20); + expect(new Set(grouped)).toEqual( + new Set([...MEASURED_ENERGY_METRIC_CONFIG_KEYS, ...POWER_BASIS_METRIC_CONFIG_KEYS]), + ); + for (const key of MEASURED_ENERGY_METRIC_CONFIG_KEYS) { + expect(getMeasuredMetricConfig(key)?.basis, key).toBe('gpu-measured'); + } expect(getMeasuredMetricConfig('y_modeledChassisPowerPerGpu')).toBeUndefined(); expect(getMeasuredMetricConfig('y_removedMetric')).toBeUndefined(); expect(getMeasuredMetricConfig('')).toBeUndefined(); @@ -42,6 +53,7 @@ describe('measured metric configuration', () => { it('keeps fleet percentiles, role averages and TDP normalization distinct', () => { expect(getMeasuredMetricConfig('y_measuredP90Power')).toEqual({ family: 'power', + basis: 'gpu-measured', scope: 'all', statistic: 'p90', display: 'watts', @@ -78,6 +90,7 @@ describe('measured metric configuration', () => { ); expect(getMeasuredMetricConfig('y_measuredPrefillJPerInputToken')).toEqual({ family: 'energy', + basis: 'gpu-measured', scope: 'prefill', denominator: 'input', unit: 'joules', diff --git a/packages/app/src/components/inference/measured-metric-config.ts b/packages/app/src/components/inference/measured-metric-config.ts index 555fbef2e..ff28546be 100644 --- a/packages/app/src/components/inference/measured-metric-config.ts +++ b/packages/app/src/components/inference/measured-metric-config.ts @@ -1,17 +1,27 @@ +import type { PowerBasis } from '@/lib/power-basis'; import type { MetricConfigKey } from './metric-registry'; export type MeasuredMetricFamily = 'power' | 'energy'; type MeasuredScope = 'all' | 'prefill' | 'decode'; +/** + * How whole-deployment average power is shown: per-chip watts, percent of + * TDP, or the per-second telemetry trace behind the average (`timeline`, which + * ChartDisplay renders with `PowerTimeline` instead of the scatter chart). + */ +export type MeasuredPowerDisplay = 'watts' | 'tdp' | 'timeline'; export type MeasuredMetricConfig = | { family: 'power'; + /** Power boundary the key plots; only `gpu-measured` publishes the other dimensions. */ + basis: PowerBasis; scope: MeasuredScope; statistic: 'average' | 'p75' | 'p90'; - display: 'watts' | 'tdp'; + display: MeasuredPowerDisplay; } | { family: 'energy'; + basis: PowerBasis; scope: MeasuredScope; denominator: 'input' | 'output' | 'total' | 'query'; unit: 'joules' | 'wattHours'; @@ -19,9 +29,10 @@ export type MeasuredMetricConfig = export type MeasuredMetricConfigChange = Partial<{ family: MeasuredMetricFamily; + basis: PowerBasis; scope: MeasuredScope; statistic: 'average' | 'p75' | 'p90'; - display: 'watts' | 'tdp'; + display: MeasuredPowerDisplay; denominator: 'input' | 'output' | 'total' | 'query'; unit: 'joules' | 'wattHours'; }>; @@ -31,51 +42,78 @@ export const MEASURED_METRIC_DEFAULTS = { energy: 'y_measuredJPerOutputToken', } as const satisfies Record; +const measured = { basis: 'gpu-measured' } as const; + // Presentation settings resolve to existing metrics; they do not own chart state. const MEASURED_METRIC_CONFIGS: readonly (readonly [MetricConfigKey, MeasuredMetricConfig])[] = [ - ['y_measuredAvgPower', { family: 'power', scope: 'all', statistic: 'average', display: 'watts' }], - ['y_measuredP75Power', { family: 'power', scope: 'all', statistic: 'p75', display: 'watts' }], - ['y_measuredP90Power', { family: 'power', scope: 'all', statistic: 'p90', display: 'watts' }], + [ + 'y_measuredAvgPower', + { family: 'power', ...measured, scope: 'all', statistic: 'average', display: 'watts' }, + ], + [ + 'y_measuredP75Power', + { family: 'power', ...measured, scope: 'all', statistic: 'p75', display: 'watts' }, + ], + [ + 'y_measuredP90Power', + { family: 'power', ...measured, scope: 'all', statistic: 'p90', display: 'watts' }, + ], [ 'y_measuredPrefillAvgPower', - { family: 'power', scope: 'prefill', statistic: 'average', display: 'watts' }, + { family: 'power', ...measured, scope: 'prefill', statistic: 'average', display: 'watts' }, ], [ 'y_measuredDecodeAvgPower', - { family: 'power', scope: 'decode', statistic: 'average', display: 'watts' }, + { family: 'power', ...measured, scope: 'decode', statistic: 'average', display: 'watts' }, ], [ 'y_measuredPowerPercentTdp', - { family: 'power', scope: 'all', statistic: 'average', display: 'tdp' }, + { family: 'power', ...measured, scope: 'all', statistic: 'average', display: 'tdp' }, + ], + [ + 'y_measuredPowerTimeline', + { family: 'power', ...measured, scope: 'all', statistic: 'average', display: 'timeline' }, ], [ 'y_measuredJPerInputToken', - { family: 'energy', scope: 'all', denominator: 'input', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'input', unit: 'joules' }, ], [ 'y_measuredJPerOutputToken', - { family: 'energy', scope: 'all', denominator: 'output', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'output', unit: 'joules' }, ], [ 'y_measuredJPerTotalToken', - { family: 'energy', scope: 'all', denominator: 'total', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'total', unit: 'joules' }, ], [ 'y_measuredPrefillJPerInputToken', - { family: 'energy', scope: 'prefill', denominator: 'input', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'prefill', denominator: 'input', unit: 'joules' }, ], [ 'y_measuredDecodeJPerOutputToken', - { family: 'energy', scope: 'decode', denominator: 'output', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'decode', denominator: 'output', unit: 'joules' }, ], [ 'y_measuredJPerSuccessfulQuery', - { family: 'energy', scope: 'all', denominator: 'query', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'query', unit: 'joules' }, ], [ 'y_measuredWhPerSuccessfulQuery', - { family: 'energy', scope: 'all', denominator: 'query', unit: 'wattHours' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'query', unit: 'wattHours' }, ], + // Derived boundaries publish one canonical combination per family: whole + // deployment, average watts, joules per output token (lib/power-basis.ts). + ...( + [ + ['gpu-provisioned', 'y_gpuProvisionedWatts', 'y_gpuProvisionedJPerOutputToken'], + ['utility-provisioned', 'y_utilityProvisionedWatts', 'y_utilityProvisionedJPerOutputToken'], + ['utility-modeled', 'y_utilityModeledWatts', 'y_utilityModeledJPerOutputToken'], + ] as const satisfies readonly (readonly [PowerBasis, MetricConfigKey, MetricConfigKey])[] + ).flatMap(([basis, watts, energy]): (readonly [MetricConfigKey, MeasuredMetricConfig])[] => [ + [watts, { family: 'power', basis, scope: 'all', statistic: 'average', display: 'watts' }], + [energy, { family: 'energy', basis, scope: 'all', denominator: 'output', unit: 'joules' }], + ]), ]; export function getMeasuredMetricConfig(metric: string): MeasuredMetricConfig | undefined { @@ -83,6 +121,8 @@ export function getMeasuredMetricConfig(metric: string): MeasuredMetricConfig | return config ? { ...config } : undefined; } +const OTHER_DIMENSIONS = ['scope', 'statistic', 'display', 'denominator', 'unit'] as const; + export function changeMeasuredMetricConfig( metric: string, change: MeasuredMetricConfigChange, @@ -93,6 +133,21 @@ export function changeMeasuredMetricConfig( current?.family === family ? current : getMeasuredMetricConfig(MEASURED_METRIC_DEFAULTS[family])!; + // Choosing a derived boundary snaps the other dimensions to its canonical + // combination. Changing any of those dimensions while on a derived boundary + // returns to GPU-measured telemetry, the only basis that publishes variants, + // so every control change lands on a real key. A family switch keeps the + // boundary: the metric key is what carries it. + const changesOtherDimension = OTHER_DIMENSIONS.some((key) => change[key] !== undefined); + const basis = + change.basis ?? (changesOtherDimension ? 'gpu-measured' : (current ?? config).basis); + if (basis !== 'gpu-measured') { + return ( + MEASURED_METRIC_CONFIGS.find( + ([, candidate]) => candidate.family === family && candidate.basis === basis, + )?.[0] ?? MEASURED_METRIC_DEFAULTS[family] + ); + } let scope = change.scope ?? config.scope; if (config.family === 'power') { @@ -103,6 +158,7 @@ export function changeMeasuredMetricConfig( MEASURED_METRIC_CONFIGS.find( ([, candidate]) => candidate.family === 'power' && + candidate.basis === 'gpu-measured' && candidate.scope === scope && candidate.statistic === statistic && candidate.display === display, @@ -123,6 +179,7 @@ export function changeMeasuredMetricConfig( MEASURED_METRIC_CONFIGS.find( ([, candidate]) => candidate.family === 'energy' && + candidate.basis === 'gpu-measured' && candidate.scope === scope && candidate.denominator === denominator && candidate.unit === unit, diff --git a/packages/app/src/components/inference/metric-registry.test.ts b/packages/app/src/components/inference/metric-registry.test.ts index 28045728e..db4011826 100644 --- a/packages/app/src/components/inference/metric-registry.test.ts +++ b/packages/app/src/components/inference/metric-registry.test.ts @@ -19,6 +19,7 @@ import { metricCostTier, metricForCostTier, metricOptionTitle, + POWER_BASIS_METRIC_CONFIG_KEYS, resolveMetricConfigKey, tokenMetricTypeForConfigKey, } from './metric-registry'; @@ -39,6 +40,10 @@ describe('metric registry', () => { expect(e2e.y_costh_roofline).toBe('lower_left'); expect(interactivity.y_measuredPowerPercentTdp_roofline).toBe('lower_right'); expect(e2e.y_measuredPowerPercentTdp_roofline).toBe('lower_left'); + for (const key of POWER_BASIS_METRIC_CONFIG_KEYS) { + expect(interactivity[`${key}_roofline`], key).toBe('lower_right'); + expect(e2e[`${key}_roofline`], key).toBe('lower_left'); + } }); it('preserves metric-specific x overrides and bilingual labels', () => { @@ -210,7 +215,11 @@ describe('metric registry', () => { ); const measuredGroup = METRIC_CONTROL_GROUPS.find((group) => group.label === 'Measured Energy'); - expect(measuredGroup?.metrics).toBe(MEASURED_ENERGY_METRIC_CONFIG_KEYS); + expect(measuredGroup?.gated).toBe(true); + expect(measuredGroup?.metrics).toEqual([ + ...MEASURED_ENERGY_METRIC_CONFIG_KEYS, + ...POWER_BASIS_METRIC_CONFIG_KEYS, + ]); }); it('classifies measured-energy config keys', () => { diff --git a/packages/app/src/components/inference/metric-registry.ts b/packages/app/src/components/inference/metric-registry.ts index ccba918b1..9687c561d 100644 --- a/packages/app/src/components/inference/metric-registry.ts +++ b/packages/app/src/components/inference/metric-registry.ts @@ -386,6 +386,80 @@ export const METRIC_REGISTRY = { titleZh: '实测平均功耗占 TDP 百分比', polarity: 'lower', }, + // The per-second telemetry behind `measuredAvgPower`. The field aliases the + // same average so the table view, availability panel, and share links keep + // working; ChartDisplay swaps the scatter chart for `PowerTimeline`, which + // fetches each point's `gpu_metrics_*` artifact and draws the trace. The + // label leads with "Measured Average Power", like the %TDP display, so a + // "Measured Power" search still finds only the family option. + measuredPowerTimeline: { + field: 'measuredPowerTimeline.y', + label: 'Measured Average Power per Chip over Time (W)', + labelZh: '每芯片实测平均功耗时间线(W)', + title: 'Measured Average Power per Chip over Time', + titleZh: '每芯片实测平均功耗时间线', + polarity: 'lower', + }, + // Power boundaries beyond GPU-measured telemetry (`lib/power-basis.ts`). + // Each boundary publishes W per allocated GPU and J per output token; the + // Boundary select in the Measured controls resolves to these keys, so the + // metric key alone carries the boundary in share links. Keys deliberately + // lack the `measured` prefix: they are spec constants or model output, not + // telemetry, so the telemetry-only decorations must not treat them as such. + gpuProvisionedWatts: { + field: 'gpuProvisionedWatts.y', + label: 'GPU Provisioned Power per Chip (TDP, W)', + labelZh: '每芯片 GPU 额定功耗(TDP,W)', + title: 'GPU Provisioned Power per Chip (TDP)', + titleZh: '每芯片 GPU 额定功耗(TDP)', + polarity: 'lower', + }, + gpuProvisionedJPerOutputToken: { + field: 'gpuProvisionedJPerOutputToken.y', + label: 'GPU Provisioned J per Output Token (TDP, J/tok)', + labelZh: '每输出 token GPU 额定能耗(TDP,J/tok)', + title: 'GPU Provisioned Joules per Output Token (TDP)', + titleZh: '每输出 token GPU 额定焦耳能耗(TDP)', + polarity: 'lower', + }, + utilityProvisionedWatts: { + field: 'utilityProvisionedWatts.y', + label: 'Utility Provisioned Power per Chip (all-in, W)', + labelZh: '每芯片全电源配置功耗(all-in,W)', + title: 'Utility Provisioned Power per Chip (all-in)', + titleZh: '每芯片全电源配置功耗(all-in)', + polarity: 'lower', + }, + // Unlike the ungated `jOutput`, which divides by output tokens per decode + // GPU, this normalizes by every allocated GPU (prefill + decode). + utilityProvisionedJPerOutputToken: { + field: 'utilityProvisionedJPerOutputToken.y', + label: 'Utility Provisioned J per Output Token, all GPUs (all-in, J/tok)', + labelZh: '每输出 token 全电源配置能耗,按全部 GPU 归一(all-in,J/tok)', + title: 'Utility Provisioned Joules per Output Token, all GPUs (all-in)', + titleZh: '每输出 token 全电源配置焦耳能耗,按全部 GPU 归一(all-in)', + polarity: 'lower', + }, + // zh vocabulary shared with the Boundary select, its help text and the chart + // caption: B3 “全电源配置” (as the ungated jOutput/jTotal already say for + // all-in), B4 “数据中心建模” (measured GPU power carried through the chassis + // model to the utility meter). + utilityModeledWatts: { + field: 'utilityModeledWatts.y', + label: 'Utility Modeled Power per Chip (PUE, W)', + labelZh: '每芯片数据中心建模功耗(含 PUE,W)', + title: 'Utility Modeled Power per Chip (PUE)', + titleZh: '每芯片数据中心建模功耗(含 PUE)', + polarity: 'lower', + }, + utilityModeledJPerOutputToken: { + field: 'utilityModeledJPerOutputToken.y', + label: 'Utility Modeled J per Output Token (PUE, J/tok)', + labelZh: '每输出 token 数据中心建模能耗(含 PUE,J/tok)', + title: 'Utility Modeled Joules per Output Token (PUE)', + titleZh: '每输出 token 数据中心建模焦耳能耗(含 PUE)', + polarity: 'lower', + }, } as const satisfies Record; export type MetricKey = keyof typeof METRIC_REGISTRY; @@ -597,6 +671,7 @@ export const MEASURED_ENERGY_METRIC_CONFIG_KEYS = [ 'y_measuredJPerSuccessfulQuery', 'y_measuredWhPerSuccessfulQuery', 'y_measuredPowerPercentTdp', + 'y_measuredPowerTimeline', ] as const satisfies readonly MetricConfigKey[]; const MEASURED_ENERGY_METRIC_CONFIG_KEY_SET: ReadonlySet = new Set( @@ -618,6 +693,31 @@ export function isRoleLocalMeasuredEnergyConfigKey(configKey: string): boolean { return ROLE_LOCAL_MEASURED_ENERGY_METRIC_CONFIG_KEY_SET.has(configKey); } +/** + * The derived power-boundary y-axes (GPU provisioned, utility provisioned, + * utility modeled) that share the gated Measured Energy group and its + * Boundary select. They are kept out of `MEASURED_ENERGY_METRIC_CONFIG_KEYS` + * on purpose: spec constants and model output carry no telemetry tier, so the + * legacy-power ring, tier tooltip line, and footer key do not apply to them. + */ +export const POWER_BASIS_METRIC_CONFIG_KEYS = [ + 'y_gpuProvisionedWatts', + 'y_gpuProvisionedJPerOutputToken', + 'y_utilityProvisionedWatts', + 'y_utilityProvisionedJPerOutputToken', + 'y_utilityModeledWatts', + 'y_utilityModeledJPerOutputToken', +] as const satisfies readonly MetricConfigKey[]; + +const POWER_BASIS_METRIC_CONFIG_KEY_SET: ReadonlySet = new Set( + POWER_BASIS_METRIC_CONFIG_KEYS, +); + +/** Whether a y-axis config key plots a derived power boundary (B2–B4). */ +export function isPowerBasisConfigKey(configKey: string): boolean { + return POWER_BASIS_METRIC_CONFIG_KEY_SET.has(configKey); +} + export const MODELED_SYSTEM_POWER_METRIC_CONFIG_KEY = 'y_modeledChassisPowerPerGpu'; /** Whether a y-axis config key plots the modeled chassis AC power metric. */ @@ -671,10 +771,13 @@ export const METRIC_CONTROL_GROUPS: readonly MetricControlGroup[] = [ // Runner power telemetry and the chassis model built on it are still being // validated, so both groups stay behind the ↑↑↓↓ feature gate until the // measurements are stable enough to publish. + // The derived boundaries ride along so the same gate and the same + // shared-URL exception (a gated metric selected by `i_metric` still renders + // while locked) apply to them. { label: 'Measured Energy', labelZh: '实测能耗', - metrics: MEASURED_ENERGY_METRIC_CONFIG_KEYS, + metrics: [...MEASURED_ENERGY_METRIC_CONFIG_KEYS, ...POWER_BASIS_METRIC_CONFIG_KEYS], gated: true, }, { diff --git a/packages/app/src/components/inference/types.ts b/packages/app/src/components/inference/types.ts index c9f9e6c96..cebfb7336 100644 --- a/packages/app/src/components/inference/types.ts +++ b/packages/app/src/components/inference/types.ts @@ -6,6 +6,7 @@ import type { Model, Sequence } from '@/lib/data-mappings'; import type { PowerTier } from '@/lib/power-tier'; import type { SystemPowerEstimate } from '@/lib/modeled-system-power'; import type { MetricKey } from './metric-registry'; +import type { PowerBasis } from '@/lib/power-basis'; export type { WorkerPower }; @@ -347,8 +348,67 @@ export interface InferenceData extends Partial void; setScaleType: (type: 'auto' | 'linear' | 'log') => void; + setPowerCompare: (mode: PowerCompare) => void; setQuickFilterVendors: (vendors: string[]) => void; setQuickFilterFrameworks: (frameworks: string[]) => void; setQuickFilterDeployment: (modes: DeploymentMode[]) => void; diff --git a/packages/app/src/components/inference/ui/ChartControls.tsx b/packages/app/src/components/inference/ui/ChartControls.tsx index 203275706..726a6918c 100644 --- a/packages/app/src/components/inference/ui/ChartControls.tsx +++ b/packages/app/src/components/inference/ui/ChartControls.tsx @@ -61,6 +61,7 @@ import { MetricExplanation } from './MetricExplanation'; import { PowerMetricAvailability } from './PowerMetricAvailability'; import { MeasuredMetricControls } from './MeasuredMetricControls'; import { + changeMeasuredMetricConfig, getMeasuredMetricConfig, MEASURED_METRIC_DEFAULTS, type MeasuredMetricFamily, @@ -230,6 +231,7 @@ export default function ChartControls({ selectedXAxisMetric, selectedXAxisMode, scaleType, + powerCompare, } = useInferenceDisplay(); const { setSelectedModel, @@ -242,6 +244,7 @@ export default function ChartControls({ setSelectedDateRange, setSelectedXAxisMetric, setScaleType, + setPowerCompare, } = useInferenceActions(); // Y-axis options come from the canonical registry and need no API data. @@ -335,10 +338,14 @@ export default function ChartControls({ if (!config) return [option]; if (seen.has(config.family)) return []; seen.add(config.family); + // Keep the selected boundary (and other dimensions) when hopping between + // the power and energy families; fall back to the family default otherwise. const value = selectedConfig?.family === config.family ? selectedYAxisMetric - : MEASURED_METRIC_DEFAULTS[config.family]; + : selectedConfig + ? changeMeasuredMetricConfig(selectedYAxisMetric, { family: config.family }) + : MEASURED_METRIC_DEFAULTS[config.family]; return [ { value, @@ -571,6 +578,8 @@ export default function ChartControls({
`vs. ${word} Time To First Token`, vsE2eLatency: (pctl?: string) => pctl ? `vs. ${pctl} End-to-end Latency` : 'vs. End-to-end Latency', @@ -166,6 +181,15 @@ const STRINGS = { noChartData: '当前模型、场景与筛选条件下没有匹配的基准测试数据。请调整上方筛选条件查看结果。', noSystemPowerData: '当前选择没有可用的系统功耗估算。请选择 8K / 1K 场景;估算仅覆盖 GPU 遥测已验证、硬件受支持、八卡机箱位置已知的运行。存在遥测数据时,仍可单独查看 GPU 实测功耗。', + noUtilityModeledData: + '当前选择没有可用的数据中心建模数值。该边界需要 8K / 1K 场景、已验证的 GPU 遥测,且硬件在机箱功耗模型覆盖范围内(不含 NVL72 系统)。可切换到其他功耗边界以保留数据点。', + powerBasisAssumptions: { + 'gpu-provisioned': + 'GPU 额定边界 · 功率取硬件注册表中每 GPU 的额定 TDP,因此每种硬件的功率曲线为水平线。每输出 token 能耗 = TDP × 分配的 GPU 数 ÷ 整个部署的输出 tok/s;分离式配置将 prefill 与 decode GPU 一并计入。未公布 TDP 的硬件不绘制。', + 'utility-provisioned': + '全电源配置边界 · 功率取硬件注册表中每 GPU 的全电源配置(all-in)市电功率(来源:SemiAnalysis Datacenter Industry Model),因此每种硬件的功率曲线为水平线。每输出 token 能耗 = all-in 功率 × 分配的 GPU 数 ÷ 整个部署的输出 tok/s;分离式配置将 prefill 与 decode GPU 一并计入,这与未加门控的“每输出 token 全电源配置能耗”按 decode GPU 计算不同。', + 'utility-modeled': `数据中心建模边界 · 将 GPU 实测功耗经机箱功耗模型(CPU、DRAM、平台开销、PSU 损耗)推算至市电侧:机箱交流功耗估算 × PUE ${AIR_COOLED_SYSTEM_PUE}(风冷,仅应用一次),再除以实测 GPU 数;每输出 token 能耗按同一比例放大实测能耗。机箱功耗模型版本 ${SYSTEM_POWER_MODEL_REVISION.slice(0, 7)}。仅适用于 8K / 1K、遥测已验证且硬件受支持的运行;NVL72 系统(GB200、GB300)及缺少数值的数据点不绘制。`, + }, vsTtft: (word: string) => `vs. ${word === 'Median' ? '中位' : word} 首 token 延迟(TTFT)`, vsE2eLatency: (pctl?: string) => (pctl ? `vs. ${pctl} 端到端延迟` : 'vs. 端到端延迟'), }, @@ -291,9 +315,19 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean selectedXAxisMode, tokenRevenuePricing, showLineLabels, + powerCompare, } = useInferenceDisplay(); const { setSelectedDates, setSelectedDatesFromRunExpansion, setIsLegendExpanded } = useInferenceActions(); + // The metric key carries the power boundary; the caption discloses it for + // the derived boundaries (there is no separate URL param). + const selectedPowerBasis = getMeasuredMetricConfig(selectedYAxisMetric)?.basis; + // The Measured Power "Timeline" display swaps the scatter body for the + // per-second telemetry traces (PowerTimeline); table view and captions are + // unchanged because the metric key aliases the measured average. + const selectedMeasuredConfig = getMeasuredMetricConfig(selectedYAxisMetric); + const isPowerTimeline = + selectedMeasuredConfig?.family === 'power' && selectedMeasuredConfig.display === 'timeline'; const selectedBenchmarkType: 'single_turn' | 'agentic_traces' = selectedSequence === Sequence.AgenticTraces ? 'agentic_traces' : 'single_turn'; const workflowInfoBenchmarkType = @@ -456,6 +490,7 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean selectedPercentile, tcoBasis, selectedXAxisMode, + powerCompare, }, ); @@ -505,6 +540,7 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean selectedXAxisMetric, selectedE2eXAxisMetric, selectedPercentile, + powerCompare, selectedXAxisMode, tokenRevenuePricing, tcoBasis, @@ -825,7 +861,9 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean

{isModeledSystemPowerConfigKey(selectedYAxisMetric) ? t.noSystemPowerData - : t.noChartData} + : selectedPowerBasis === 'utility-modeled' + ? t.noUtilityModeledData + : t.noChartData}

, ] @@ -1013,6 +1051,8 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean {getSequenceLabel(graph.sequence as Sequence, locale)}{' '} {metricChartTitle(graph.chartDefinition, selectedYAxisMetric, locale)}{' '} {(() => { + // The timeline's x axis is time, not the scatter x metric. + if (isPowerTimeline) return null; const xField = graph.chartDefinition.x_scale_field; if (xField?.endsWith('_ttft')) { const percentile = xField.replace(/_ttft$/u, ''); @@ -1140,6 +1180,15 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean {t.systemPowerAssumptions}

)} + {selectedPowerBasis && selectedPowerBasis !== 'gpu-measured' && ( +

+ {t.powerBasisAssumptions[selectedPowerBasis]} +

+ )} {isUnofficialRun && selectedXAxisMode === 'e2e-normalized-interactivity' && (

@@ -1189,11 +1238,42 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean ); } + if (isPowerTimeline) { + return ( +

+ entry.point), + ]} + overlayData={ + selectUnofficialOverlayForMode( + selectedXAxisMode, + graph.chartDefinition.chartType, + overlayDataByChartType, + ) ?? undefined + } + yLabel={metricLabel( + graph.chartDefinition, + selectedYAxisMetric, + locale, + )} + caption={chartCaption} + /> +
+ ); + } + return isGpuComparison ? ( !point.powerVariant)} xLabel={resolvedXLabel} yLabel={metricLabel(graph.chartDefinition, selectedYAxisMetric, locale)} chartDefinition={graph.chartDefinition} diff --git a/packages/app/src/components/inference/ui/GPUGraph.tsx b/packages/app/src/components/inference/ui/GPUGraph.tsx index 01fc782eb..ac17588d9 100644 --- a/packages/app/src/components/inference/ui/GPUGraph.tsx +++ b/packages/app/src/components/inference/ui/GPUGraph.tsx @@ -1,5 +1,7 @@ 'use client'; +import { useFeatureGate } from '@/lib/use-feature-gate'; +import { getMeasuredMetricConfig } from '@/components/inference/measured-metric-config'; import { track } from '@/lib/analytics'; import { isPersistedBenchmarkId } from '@/lib/benchmark-id'; import { useEphemeralUrlState } from '@/hooks/useUrlState'; @@ -48,6 +50,7 @@ import { chartFrontier, upperPowerEnvelope, isPowerCurveMetric, + isPowerGaugeSeries, isMeasuredPowerCurveMetric, } from '@/components/inference/utils/powerCurves'; import type { @@ -109,6 +112,14 @@ import { } from '@/components/inference/ui/line-label-layer'; import { QuickFiltersDialog } from '@/components/inference/ui/QuickFiltersDialog'; +const PowerTelemetryDialog = dynamic( + () => + import('@/components/inference/power-telemetry-dialog').then( + (module) => module.PowerTelemetryDialog, + ), + { ssr: false }, +); + const FixedSequenceLogDialog = dynamic(() => import('@/components/inference/log-viewer/fixed-sequence-log-dialog').then( (module) => module.FixedSequenceLogDialog, @@ -255,6 +266,11 @@ const GPUGraph = React.memo( setQuickFilterPower, } = useInferenceActions(); const locale = useLocale(); + const featureGateUnlocked = useFeatureGate(); + const showPowerTelemetry = + featureGateUnlocked || getMeasuredMetricConfig(selectedYAxisMetric) !== undefined; + const showPowerTelemetryRef = useRef(showPowerTelemetry); + showPowerTelemetryRef.current = showPowerTelemetry; const legendT = GPU_STRINGS[locale]; const frontierDirection = chartDefinition[ `${selectedYAxisMetric}_roofline` as keyof ChartDefinition @@ -443,10 +459,20 @@ const GPUGraph = React.memo( if (!powerEnvelopeMode) return paretoRooflines; const result: Record = {}; for (const [key, points] of Object.entries(groupedData)) { - result[key] = upperPowerEnvelope(points, chartDefinition.chartType !== 'e2e'); + result[key] = upperPowerEnvelope( + points, + chartDefinition.chartType !== 'e2e', + isPowerGaugeSeries(selectedYAxisMetric, points[0]), + ); } return result; - }, [powerEnvelopeMode, groupedData, paretoRooflines, chartDefinition.chartType]); + }, [ + powerEnvelopeMode, + groupedData, + paretoRooflines, + chartDefinition.chartType, + selectedYAxisMetric, + ]); const boundaryPointKeys = useMemo(() => { const keys = new Set(); @@ -511,6 +537,7 @@ const GPUGraph = React.memo( const logAvailabilityRef = useRef(logAvailability); logAvailabilityRef.current = logAvailability; const [fixedLogPointId, setFixedLogPointId] = useState(null); + const [powerTelemetryPoint, setPowerTelemetryPoint] = useState(null); // Warning annotations for visible series with known upstream issues — // same treatment the scatter view gets, applied to the date-comparison view. @@ -1251,7 +1278,7 @@ const GPUGraph = React.memo( ); } - return ( + const chart = ( ref={chartRef} // Embeds drop the zoom/pan hint line; the host page has its own caption. @@ -1363,6 +1390,7 @@ const GPUGraph = React.memo( yLabel, selectedYAxisMetric, hardwareConfig, + showPowerTelemetry: showPowerTelemetryRef.current, runUrl: d.run_url ? updateRepoUrl(d.run_url) : undefined, hasTrace: isPersistedBenchmarkId(d.id) ? traceAvailabilityRef.current?.[d.id] === true @@ -1424,6 +1452,19 @@ const GPUGraph = React.memo( }); }); } + const powerBtn = tooltipEl.querySelector('[data-action="view-power-telemetry"]'); + if (powerBtn && isPersistedBenchmarkId(d.id)) { + powerBtn.addEventListener('click', (event) => { + event.stopPropagation(); + setPowerTelemetryPoint(d); + chartRef.current?.dismissTooltip(); + track('inference_power_telemetry_opened', { + id: d.id, + hwKey: d.hwKey, + conc: d.conc, + }); + }); + } const logsBtn = tooltipEl.querySelector('[data-action="view-logs"]'); if (logsBtn && typeof d.id === 'number') { logsBtn.addEventListener('click', (event) => { @@ -1694,6 +1735,21 @@ const GPUGraph = React.memo( } /> ); + + return ( + <> + {powerTelemetryPoint === null ? null : ( + { + if (!open) setPowerTelemetryPoint(null); + }} + /> + )} + {chart} + + ); }, ); diff --git a/packages/app/src/components/inference/ui/InferenceTable.test.ts b/packages/app/src/components/inference/ui/InferenceTable.test.ts index bf9ab1b0c..09294dddc 100644 --- a/packages/app/src/components/inference/ui/InferenceTable.test.ts +++ b/packages/app/src/components/inference/ui/InferenceTable.test.ts @@ -1,7 +1,12 @@ import { describe, it, expect } from 'vitest'; +import { createElement } from 'react'; +import { renderToStaticMarkup } from 'react-dom/server'; import type { ChartDefinition, InferenceData } from '@/components/inference/types'; -import { formatInferenceTableNumber } from '@/components/inference/ui/InferenceTable'; +import InferenceTable, { + formatInferenceTableNumber, +} from '@/components/inference/ui/InferenceTable'; +import { expandPowerCompareSeries } from '../utils/power-compare'; import * as inferenceTableModule from './InferenceTable'; import { chartDefinitions } from '../metric-registry'; @@ -42,6 +47,54 @@ function makePoint(overrides: Partial): InferenceData { } describe('InferenceTable sorting logic', () => { + it.each(['roles', 'boundaries'] as const)( + 'renders and sorts each %s comparison by its plotted value', + (mode) => { + const base = makePoint({ + hwKey: 'gb300_dynamo-trt', + y: 708.1, + measuredAvgPower: { y: 708.1, roof: false }, + measuredPrefillAvgPower: { y: 760.442, roof: false }, + measuredDecodeAvgPower: { y: 690.652, roof: false }, + gpuProvisionedWatts: { y: 1400, roof: false }, + utilityProvisionedWatts: { y: 1920, roof: false }, + }); + const points = expandPowerCompareSeries([base], 'y_measuredAvgPower', mode); + const sorted = sortRowsByYMetric(points, chartDefinitions[0], 'y_measuredAvgPower'); + expect(sorted.map((point) => point.y)).toEqual( + mode === 'roles' ? [690.652, 708.1, 760.442] : [708.1, 1400, 1920], + ); + + const html = renderToStaticMarkup( + createElement(InferenceTable, { + data: points, + chartDefinition: chartDefinitions[0], + selectedYAxisMetric: 'y_measuredAvgPower', + }), + ); + const body = html.split('')[1].split('')[0]; + const cells = [...body.matchAll(/]*>(?.*?)<\/tr>/gu)].map((match) => + [...match.groups!.row.matchAll(/]*>(?.*?)<\/td>/gu)].map( + (cell) => cell.groups!.cell, + ), + ); + expect(cells.map((row) => [row[2], row[3]])).toEqual( + mode === 'roles' + ? [ + ['Decode GPUs', '691'], + ['All GPUs', '708'], + ['Prefill GPUs', '760'], + ] + : [ + ['GPU measured', '708'], + ['GPU provisioned (TDP)', '1,400'], + ['Utility provisioned (all-in)', '1,920'], + ], + ); + expect(points.every((point) => point.measuredAvgPower?.y === 708.1)).toBe(true); + }, + ); + it('sorts supported modeled estimates by ascending power', () => { const definition = chartDefinitions[0]; const metric = 'y_modeledChassisPowerPerGpu'; diff --git a/packages/app/src/components/inference/ui/InferenceTable.tsx b/packages/app/src/components/inference/ui/InferenceTable.tsx index b6cdafa7e..0d4e8ab74 100644 --- a/packages/app/src/components/inference/ui/InferenceTable.tsx +++ b/packages/app/src/components/inference/ui/InferenceTable.tsx @@ -7,6 +7,7 @@ import { type DataTableColumn, DataTable } from '@/components/ui/data-table'; import { chipCounts } from '@/lib/chip-counts'; import { getNestedYValue, metricLabel, xAxisLabel } from '@/lib/chart-utils'; import { isModeledSystemPowerConfigKey } from '@/components/inference/metric-registry'; +import { inferPowerCompare, powerSeriesLabel } from '@/components/inference/utils/power-compare'; import { sortRowsByYMetric } from '@/components/inference/ui/inference-table-sort'; import { type Precision, getPrecisionLabel } from '@/lib/data-mappings'; import { getDisplayLabel } from '@/lib/utils'; @@ -43,6 +44,7 @@ export function inferenceTableHeaderLabels( physicalChips: locale === 'zh' ? '物理芯片数' : 'Physical Chips', configuredChips: locale === 'zh' ? '配置中的芯片数' : 'Configured Chip Count', concurrency: locale === 'zh' ? '并发数' : 'Conc', + series: locale === 'zh' ? '系列' : 'Series', yMetric: metricLabel(chartDefinition, selectedYAxisMetric, locale), xMetric: xAxisLabel(chartDefinition, locale), throughput: locale === 'zh' ? '单芯片吞吐量 (tok/s)' : 'Throughput/Chip (tok/s)', @@ -66,6 +68,9 @@ export default function InferenceTable({ () => sortRowsByYMetric(data, chartDefinition, selectedYAxisMetric), [data, chartDefinition, selectedYAxisMetric], ); + // Boundary / role clones (`i_pcompare`) share every config column with their + // base row; the series column is what tells them apart. + const powerCompare = useMemo(() => inferPowerCompare(data), [data]); const columns = useMemo[]>( () => [ @@ -85,6 +90,19 @@ export default function InferenceTable({ className: 'whitespace-nowrap', importance: 'key', }, + ...(powerCompare === 'none' + ? [] + : [ + { + header: headers.series, + cell: (row: InferenceData) => + powerSeriesLabel(row, selectedYAxisMetric, powerCompare, locale), + sortValue: (row: InferenceData) => + powerSeriesLabel(row, selectedYAxisMetric, powerCompare, locale), + className: 'whitespace-nowrap', + importance: 'key' as const, + }, + ]), { header: headers.tensorParallelism, align: 'right', @@ -129,8 +147,12 @@ export default function InferenceTable({ { header: headers.yMetric, align: 'right', - cell: (row) => formatInferenceTableNumber(yPath ? getNestedYValue(row, yPath) : row.y), - sortValue: (row) => (yPath ? getNestedYValue(row, yPath) : row.y), + // Comparison clones keep the source metrics; y holds the plotted role/boundary. + cell: (row) => + formatInferenceTableNumber( + row.powerVariant || !yPath ? row.y : getNestedYValue(row, yPath), + ), + sortValue: (row) => (row.powerVariant || !yPath ? row.y : getNestedYValue(row, yPath)), className: 'tabular-nums', importance: 'key', }, @@ -151,7 +173,7 @@ export default function InferenceTable({ importance: 'key', }, ], - [yPath, headers, showModeledPower], + [yPath, headers, showModeledPower, powerCompare, selectedYAxisMetric, locale], ); return ( diff --git a/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx b/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx index 46f0600eb..0373fc5bd 100644 --- a/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx +++ b/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx @@ -7,15 +7,24 @@ import { SelectTrigger, SelectValue, } from '@/components/ui/select'; +import { track } from '@/lib/analytics'; +import { POWER_BASES, POWER_BASIS_LABELS, type PowerBasis } from '@/lib/power-basis'; import { useLocale } from '@/lib/use-locale'; import { changeMeasuredMetricConfig, getMeasuredMetricConfig, type MeasuredMetricConfigChange, } from '../measured-metric-config'; +import type { PowerCompare } from '../types'; +import { POWER_COMPARE_MODES, powerCompareAvailable } from '../utils/power-compare'; const STRINGS = { en: { + basis: 'Boundary', + basisHelp: + 'Where power is counted. GPU measured: runner telemetry from the GPU boards. GPU provisioned: rated TDP per GPU. Utility provisioned: all-in provisioned utility power per GPU. Utility modeled: measured GPU power carried through the modeled chassis to the utility meter with PUE. Points without a value for the chosen boundary are omitted, never replaced with an estimate.', + basisHint: + 'Derived boundaries report average power per chip across all GPUs, and whole-deployment joules per output token. Changing another setting returns to GPU measured.', scope: 'Scope', scopeHelp: 'All GPUs measures the whole deployment. Prefill and decode select only GPUs serving that role.', @@ -29,7 +38,8 @@ const STRINGS = { roleHint: 'Prefill and decode power support Average only.', display: 'Display', displayHelp: - 'Power per chip in watts, or average power as a percentage of chip TDP. Percent of TDP is available for the all-GPU average only.', + 'Power per chip in watts, average power as a percentage of chip TDP, or the per-second telemetry timeline behind the average. Percent of TDP and Timeline are available for the all-GPU average only.', + timeline: 'Timeline', denominator: 'Per', denominatorHelp: 'Choose the energy denominator. All-GPU energy per input or output token includes the whole deployment; role energy is selected separately under Scope.', @@ -40,8 +50,21 @@ const STRINGS = { unit: 'Unit', unitHelp: 'Energy is shown in joules. Energy per successful query can also be shown in watt-hours.', + compare: 'Compare', + compareHelp: + 'Overlay sibling series on the same points, in the hardware colour with a dash per series. All boundaries: GPU measured, GPU provisioned, utility provisioned and utility modeled. Prefill vs decode: each worker pool next to the whole deployment; on the energy axis the prefill pool is carried onto the output-token axis by the served input:output ratio. Available for the whole-deployment average W/chip and J per output token.', + compareNone: 'Off', + compareBoundaries: 'All boundaries', + compareRoles: 'Prefill vs decode', + compareUnavailable: + 'The comparison is paused for this setting: it needs the whole-deployment average W/chip or J per output token.', }, zh: { + basis: '功耗边界', + basisHelp: + '选择功耗的计量边界。GPU 实测:来自 GPU 板卡的运行器遥测;GPU 额定:每 GPU 的额定 TDP;全电源配置:每 GPU 的全电源配置(all-in)市电功率;数据中心建模:将 GPU 实测功耗经机箱功耗模型推算至市电侧并计入 PUE。所选边界缺少数值的数据点将被省略,不会用估算值替代。', + basisHint: + '推导边界提供全部 GPU 的平均每芯片功率,以及整个部署的每输出 token 能耗;更改其他设置将返回 GPU 实测。', scope: '统计范围', scopeHelp: '全部 GPU 对应整个部署;预填充和解码仅统计承担相应任务的 GPU。', all: '全部 GPU', @@ -54,7 +77,8 @@ const STRINGS = { roleHint: '预填充和解码功率仅支持平均值。', display: '显示方式', displayHelp: - '显示单芯片功率(瓦),或平均功率占芯片 TDP 的百分比。TDP 百分比仅支持全部 GPU 的平均功率。', + '显示单芯片功率(瓦)、平均功率占芯片 TDP 的百分比,或平均值背后的逐秒遥测时间线。TDP 百分比和时间线仅支持全部 GPU 的平均功率。', + timeline: '时间线', denominator: '能耗分母', denominatorHelp: '选择能耗的分母。按输入或输出 token 归一化的全部 GPU 能耗仍包含整个部署;预填充或解码能耗需在统计范围中单独选择。', @@ -64,22 +88,44 @@ const STRINGS = { query: '成功请求', unit: '单位', unitHelp: '能耗以焦耳显示;每个成功请求的能耗也可显示为瓦时。', + compare: '对比', + compareHelp: + '在同一批数据点上叠加同源系列:颜色仍按硬件区分,每个系列用不同虚线表示。全部边界:GPU 实测、GPU 额定、全电源配置、数据中心建模;预填充 vs 解码:各 worker 池与整个部署并列,能耗轴上的预填充能耗按实际服务的输入/输出 token 比折算到每输出 token。仅适用于整个部署的平均 W/芯片和每输出 token 能耗。', + compareNone: '关闭', + compareBoundaries: '全部边界', + compareRoles: '预填充 vs 解码', + compareUnavailable: '当前设置下对比已暂停:需要整个部署的平均 W/芯片或每输出 token 能耗。', }, } as const; export function MeasuredMetricControls({ metric, onChange, + compare = 'none', + onCompareChange, }: { metric: string; onChange: (metric: string) => void; + /** Comparison series overlaid on the metric (`i_pcompare`). */ + compare?: PowerCompare; + onCompareChange?: (mode: PowerCompare) => void; }) { - const t = STRINGS[useLocale()]; + const locale = useLocale(); + const t = STRINGS[locale]; const config = getMeasuredMetricConfig(metric); if (!config) return null; + const compareLabels: Record = { + none: t.compareNone, + boundaries: t.compareBoundaries, + roles: t.compareRoles, + }; + const compareActive = compare !== 'none'; + const compareApplies = powerCompareAvailable(metric, compare); const change = (next: MeasuredMetricConfigChange) => onChange(changeMeasuredMetricConfig(metric, next)); + const basisId = `measured-${config.family}-basis`; const scopeId = `measured-${config.family}-scope`; + const derivedBasis = config.basis !== 'gpu-measured'; const roleScope = config.family === 'energy' ? config.denominator === 'input' @@ -91,9 +137,31 @@ export function MeasuredMetricControls({ return (
+
+ + +
{config.family === 'energy' && (
change({ statistic })} @@ -204,6 +272,13 @@ export function MeasuredMetricControls({ > % TDP + + {t.timeline} +
@@ -237,6 +312,59 @@ export function MeasuredMetricControls({
)} + {onCompareChange && ( +
+ + +
+ )} + {derivedBasis && ( +

+ {t.basisHint} +

+ )} + {compareActive && !compareApplies && ( +

+ {t.compareUnavailable} +

+ )}
); } diff --git a/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx b/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx index a3dd548b0..cad29732b 100644 --- a/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx +++ b/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx @@ -2,8 +2,16 @@ import { useMemo } from 'react'; import { useInferenceData, useInferenceFilters } from '../InferenceContext'; -import { isMeasuredEnergyConfigKey, metricOptionTitle, type MetricKey } from '../metric-registry'; +import { + isMeasuredEnergyConfigKey, + isPowerBasisConfigKey, + metricOptionTitle, + POWER_BASIS_METRIC_CONFIG_KEYS, + type MetricKey, +} from '../metric-registry'; +import { getMeasuredMetricConfig } from '../measured-metric-config'; import type { InferenceData } from '../types'; +import { powerBasisNormalization } from '@/lib/power-basis'; import { matchesQuickFilters } from '../utils/quickFilters'; import { powerMetricAvailability, @@ -33,6 +41,17 @@ const STRINGS = { ambiguous: 'Whole-deployment energy schema unavailable', missing: 'Metric not reported', }, + basisLabels: { + available: 'Value available', + noSpec: 'No published spec for this hardware', + noThroughput: 'No output throughput reported', + noNormalization: 'Whole-deployment GPU count unavailable for this disaggregated row', + noTelemetry: 'No validated GPU telemetry', + invalid: 'Validation failed', + modelWorkload: 'Chassis model covers 8K / 1K only', + modelHardware: 'Hardware not in the chassis power model', + modelUnsupported: 'Chassis model unsupported for this deployment', + }, note: 'A missing verdict does not establish age or validity. Prefill/decode metrics measure separate worker pools. Missing values are never replaced with zero or TDP estimates.', all: 'Availability of all measured metrics', evidence: 'Selected metric: source details', @@ -54,6 +73,17 @@ const STRINGS = { ambiguous: '缺少整个部署的能耗 schema', missing: '未提供此指标', }, + basisLabels: { + available: '有数值', + noSpec: '该硬件没有公开的规格参数', + noThroughput: '未报告输出吞吐量', + noNormalization: '无法确定该分离式部署的 GPU 总数', + noTelemetry: '没有已验证的 GPU 遥测', + invalid: '验证失败', + modelWorkload: '机箱功耗模型仅覆盖 8K / 1K', + modelHardware: '硬件不在机箱功耗模型范围内', + modelUnsupported: '机箱功耗模型不支持此部署', + }, note: '缺少验证结论不能判断数据新旧或有效性。prefill/decode 指标仅衡量独立 worker 池。缺失值不会被替换为零或 TDP 估算值。', all: '所有实测指标的可用性', evidence: '当前指标的来源详情', @@ -61,6 +91,99 @@ const STRINGS = { }, } as const; +/** + * Why a point lacks a derived power boundary (lib/power-basis.ts). Provisioned + * boundaries are spec constants, so their watts exist for any registered + * hardware and their energy additionally needs output throughput plus, for + * disaggregated rows, the whole-deployment GPU count that + * `powerBasisNormalization` recovers only for fixed-sequence runs with integer + * prefill/decode counts. The modeled boundary is measured telemetry carried + * through the chassis model, so it inherits the telemetry verdict and the + * model's own unsupported reasons. + */ +export type PowerBasisAvailabilityState = + | 'available' + | 'noSpec' + | 'noThroughput' + | 'noNormalization' + | 'noTelemetry' + | 'invalid' + | 'modelWorkload' + | 'modelHardware' + | 'modelUnsupported'; + +const POWER_BASIS_AVAILABILITY_STATES: readonly PowerBasisAvailabilityState[] = [ + 'available', + 'noSpec', + 'noThroughput', + 'noNormalization', + 'noTelemetry', + 'invalid', + 'modelWorkload', + 'modelHardware', + 'modelUnsupported', +]; + +const hasFiniteValue = (point: InferenceData, key: MetricKey): boolean => { + const value = point[key]; + return ( + typeof value === 'object' && + value !== null && + 'y' in value && + typeof value.y === 'number' && + Number.isFinite(value.y) + ); +}; + +export function powerBasisState( + point: InferenceData, + configKey: string, +): PowerBasisAvailabilityState { + const key = configKey.replace(/^y_/u, '') as MetricKey; + if (hasFiniteValue(point, key)) return 'available'; + const config = getMeasuredMetricConfig(configKey); + if (config?.basis === 'utility-modeled') { + if (point.power_valid === 0) return 'invalid'; + const model = point.modeledSystemPower; + if (model?.status === 'unsupported') { + if (model.reason === 'workload') return 'modelWorkload'; + if (model.reason === 'hardware') return 'modelHardware'; + if (model.reason === 'telemetry') return 'noTelemetry'; + return 'modelUnsupported'; + } + // A supported model without a plotted value means B1 is absent (B4 follows B1). + return 'noTelemetry'; + } + // Provisioned energy needs the watts sibling plus the whole-deployment + // normalization; the same helper that withheld the value says which half is + // missing, so the explanation cannot drift from the formula. + const wattsKey = ( + config?.basis === 'gpu-provisioned' ? 'gpuProvisionedWatts' : 'utilityProvisionedWatts' + ) satisfies MetricKey; + if (!hasFiniteValue(point, wattsKey)) return 'noSpec'; + const perGpu = point.output_tput_per_gpu; + if (typeof perGpu !== 'number' || !Number.isFinite(perGpu) || perGpu <= 0) return 'noThroughput'; + // Throughput exists, so only the disaggregated GPU count can be missing. + // Chart points may lack the counts an aggregate entry always has; an + // unknown count is exactly the "unavailable" case the helper reports. + const { allocatedGpus } = powerBasisNormalization({ + output_tput_per_gpu: perGpu, + disagg: point.disagg ?? false, + benchmark_type: point.benchmark_type, + num_prefill_gpu: point.num_prefill_gpu ?? Number.NaN, + num_decode_gpu: point.num_decode_gpu ?? Number.NaN, + }); + return allocatedGpus === null ? 'noNormalization' : 'noThroughput'; +} + +function powerBasisAvailability(points: readonly InferenceData[], metric: string) { + const counts = Object.fromEntries( + POWER_BASIS_AVAILABILITY_STATES.map((state) => [state, 0]), + ) as Record; + for (const point of points) counts[powerBasisState(point, metric)]++; + return { metric, counts, available: counts.available, total: points.length }; +} + export function PowerMetricAvailabilityPanel({ points, metric, @@ -74,16 +197,35 @@ export function PowerMetricAvailabilityPanel({ }) { const locale = useLocale(); const t = STRINGS[locale]; - const availability = useMemo(() => powerMetricAvailability(points), [points]); + const availability = useMemo( + () => [ + ...powerMetricAvailability(points), + ...POWER_BASIS_METRIC_CONFIG_KEYS.map((key) => powerBasisAvailability(points, key)), + ], + [points], + ); const selected = availability.find((entry) => entry.metric === metric); if (!selected) return null; + const isBasis = isPowerBasisConfigKey(metric); + const stateOf = (point: InferenceData) => + isBasis ? powerBasisState(point, metric) : powerMetricState(point, metric); + // The two dictionaries overlap on `invalid`; the selected metric, not the + // key, decides which copy applies so the measured strings stay untouched. + const labelOf = (state: PowerAvailabilityState | PowerBasisAvailabilityState) => + isBasis + ? t.basisLabels[state as PowerBasisAvailabilityState] + : t.labels[state as PowerAvailabilityState]; const sources = new Map< string, - { point: InferenceData; state: PowerAvailabilityState; count: number } + { + point: InferenceData; + state: PowerAvailabilityState | PowerBasisAvailabilityState; + count: number; + } >(); for (const point of points) { - const state = powerMetricState(point, metric); - if (state === 'strict') continue; + const state = stateOf(point); + if (state === 'strict' || state === 'available') continue; const key = JSON.stringify([point.hwKey, point.run_url, state, point.power_invalid_reasons]); const group = sources.get(key); if (group) group.count++; @@ -108,7 +250,10 @@ export function PowerMetricAvailabilityPanel({

{Object.entries(selected.counts) .filter(([, count]) => count > 0) - .map(([state, count]) => `${t.labels[state as PowerAvailabilityState]}: ${count}`) + .map( + ([state, count]) => + `${labelOf(state as PowerAvailabilityState | PowerBasisAvailabilityState)}: ${count}`, + ) .join(' · ')}

{[...sources.values()].map(({ point, state, count }, index) => (
  • - {point.hwKey}: {t.labels[state]} ({count}) + {point.hwKey}: {labelOf(state)} ({count}) {point.power_invalid_reasons?.length ? ` · ${point.power_invalid_reasons.join(', ')}` : ''} @@ -221,7 +366,7 @@ export function PowerMetricAvailability({ quickFilters, compareGpuPair, ]); - if (!isMeasuredEnergyConfigKey(metric)) return null; + if (!isMeasuredEnergyConfigKey(metric) && !isPowerBasisConfigKey(metric)) return null; return ( vi.mocked(setupChartStructure).mock.calls.leng beforeEach(() => { globalThis.__scatterPathnameState.value = '/inference'; vi.stubGlobal('ResizeObserver', MockResizeObserver); + Object.defineProperty(SVGElement.prototype, 'getComputedTextLength', { + configurable: true, + value(this: SVGElement) { + return (this.textContent?.length ?? 0) * 7; + }, + }); Object.defineProperty(SVGElement.prototype, 'getBBox', { configurable: true, value: () => @@ -259,6 +269,15 @@ beforeEach(() => { afterEach(() => { vi.unstubAllGlobals(); vi.restoreAllMocks(); + if (originalGetComputedTextLength) { + Object.defineProperty( + SVGElement.prototype, + 'getComputedTextLength', + originalGetComputedTextLength, + ); + } else { + Reflect.deleteProperty(SVGElement.prototype, 'getComputedTextLength'); + } if (originalGetBBox) { Object.defineProperty(SVGElement.prototype, 'getBBox', originalGetBBox); } else { diff --git a/packages/app/src/components/inference/ui/ScatterGraph.tsx b/packages/app/src/components/inference/ui/ScatterGraph.tsx index fc860d3ad..9f6656adf 100644 --- a/packages/app/src/components/inference/ui/ScatterGraph.tsx +++ b/packages/app/src/components/inference/ui/ScatterGraph.tsx @@ -1,10 +1,13 @@ 'use client'; +import { useFeatureGate } from '@/lib/use-feature-gate'; +import { getMeasuredMetricConfig } from '@/components/inference/measured-metric-config'; import { track } from '@/lib/analytics'; import { isPersistedBenchmarkId } from '@/lib/benchmark-id'; import { useEphemeralUrlState } from '@/hooks/useUrlState'; import { rememberChartStateInUrl } from '@/lib/url-state'; import * as d3 from 'd3'; +import { CHART_TYPE } from '@/lib/d3-chart/typography'; import dynamic from 'next/dynamic'; import React, { useCallback, useEffect, useLayoutEffect, useMemo, useRef, useState } from 'react'; @@ -16,6 +19,7 @@ import { useInferenceDisplay, useInferenceFilters, } from '@/components/inference/InferenceContext'; +import { usePerfRulerStore } from '@/components/inference/perf-ruler-store'; import { useTraceAvailability } from '@/hooks/api/use-trace-availability'; import { useLogAvailability } from '@/hooks/api/use-log-availability'; import { computeToggle } from '@/hooks/useTogglableSet'; @@ -47,7 +51,13 @@ import { matchKnownConfigIssues, pointMatchesIssue } from '@/lib/known-issues'; import { useLocale } from '@/lib/use-locale'; import { getLineLabelVendorIcon } from '@/lib/vendor-logos'; import { formatNumber, getDisplayLabel, updateRepoUrl } from '@/lib/utils'; -import { getInferenceHardwareConfig, getInferenceRunLabel } from '@/lib/inference-labels'; +import { + getInferenceHardwareConfig, + getInferenceRunLabel, + getOverlayLineLabel, + OVERLAY_LABEL_MARKER, + overlayRunTag, +} from '@/lib/inference-labels'; import { D3Chart } from '@/lib/d3-chart/D3Chart'; import type { CustomLayerConfig, @@ -76,6 +86,7 @@ import { renderPerfRulers, type PerfRulerEndInput, type PerfRulerGeometry, + type PerfRulerMeasurement, type PerfRulerRenderEntry, type PerfRulerState, } from '@/lib/d3-chart/layers/perf-ruler'; @@ -111,6 +122,7 @@ import { chartFrontier, upperPowerEnvelope, isPowerCurveMetric, + isPowerGaugeSeries, isMeasuredPowerCurveMetric, } from '@/components/inference/utils/powerCurves'; import type { @@ -118,17 +130,38 @@ import type { ClippedInferenceData, InferenceData, ScatterGraphProps, + PowerVariant, } from '@/components/inference/types'; import { generateOverlayTooltipContent, generateTooltipContent, } from '@/components/inference/utils/tooltipUtils'; +import { + POWER_TIMELINE_METRIC_KEY, + requestPowerTraceFocus, + traceKeyForPoint, +} from '@/components/inference/utils/powerTimeline'; import { QuickFiltersDialog } from '@/components/inference/ui/QuickFiltersDialog'; import { ScatterEmptyState } from '@/components/inference/ui/ScatterEmptyState'; import { scatterPointConfigId, scatterPointJoinId, + parseScatterSeriesKey, + scatterSeriesKey, } from '@/components/inference/utils/point-identity'; +import { + flatSeriesValue, + inferPowerCompare, + lineLabelHardwareKey, + lineLabelSeriesId, + metricPlotsWatts, + powerCompareBase, + powerLineLabelSuffix, + powerVariantDash, + powerVariantId, + powerVariantLabel, + powerVariantsInData, +} from '@/components/inference/utils/power-compare'; import LegendPointsDialog from '@/components/inference/ui/LegendPointsDialog'; import { renderOffloadHalo } from '@/components/inference/utils/offload-halo'; import { renderLegacyPowerRing } from '@/components/inference/utils/legacy-power-marker'; @@ -163,6 +196,14 @@ import { fitContinuationLabelBaseline, } from '@/components/inference/utils/overflowContinuations'; +const PowerTelemetryDialog = dynamic( + () => + import('@/components/inference/power-telemetry-dialog').then( + (module) => module.PowerTelemetryDialog, + ), + { ssr: false }, +); + const FixedSequenceLogDialog = dynamic( () => import('@/components/inference/log-viewer/fixed-sequence-log-dialog').then( @@ -221,6 +262,32 @@ const optimalPointKey = (d: InferenceData): string => const EMPTY_OVERLAY_DATA: InferenceData[] = []; const EMPTY_CLIPPED_DATA: ClippedInferenceData[] = []; +/** + * Legend ids of the comparison-series rows (`i_pcompare`), distinct from + * hardware keys so the shared hover / toggle handlers can tell them apart. + */ +const POWER_VARIANT_LEGEND_PREFIX = 'power-variant:'; +/** Comparison clones sit behind the base series they annotate. */ +const POWER_VARIANT_POINT_OPACITY = 0.6; +const pointOpacityForVariant = (d: InferenceData): number => + d.powerVariant ? POWER_VARIANT_POINT_OPACITY : 1; +/** Dash for a series key's variant id (`parseScatterSeriesKey().variant`). */ +const powerVariantDashById = (variantId: string | null | undefined): string => + variantId ? (VARIANT_DASH_BY_ID.get(variantId) ?? '') : ''; +const VARIANT_DASH_BY_ID = new Map( + ( + [ + ['basis', 'gpu-measured'], + ['basis', 'gpu-provisioned'], + ['basis', 'utility-provisioned'], + ['basis', 'utility-modeled'], + ['role', 'all'], + ['role', 'prefill'], + ['role', 'decode'], + ] as const + ).map(([kind, id]) => [id, powerVariantDash({ kind, id } as PowerVariant)]), +); + const LINE_LABEL_RAISE = ['.line-label'] as const; /** Decorations sit above the visible shape, which a precision toggle may replace. */ const POINT_DECORATION_RAISE = ['.offload-halo', '.legacy-power-ring'] as const; @@ -534,6 +601,7 @@ const ScatterGraph = React.memo( setQuickFilterDeployment, setQuickFilterSpec, setQuickFilterPower, + setSelectedYAxisMetric, } = useInferenceActions(); const paretoDirection = chartDefinition[`${selectedYAxisMetric}_roofline`] as | ParetoDirection @@ -552,15 +620,31 @@ const ScatterGraph = React.memo( const groups = groupPointsByDate(points); if (showPowerEnvelope) { for (const [date, samples] of groups) { - groups.set(date, upperPowerEnvelope(samples, chartDefinition.chartType !== 'e2e')); + groups.set( + date, + upperPowerEnvelope( + samples, + chartDefinition.chartType !== 'e2e', + isPowerGaugeSeries(selectedYAxisMetric, samples[0]), + ), + ); } } return groups; }, - [showPowerEnvelope, chartDefinition.chartType], + [showPowerEnvelope, chartDefinition.chartType, selectedYAxisMetric], ); const locale = useLocale(); + const featureGateUnlocked = useFeatureGate(); + const showPowerTelemetry = + featureGateUnlocked || getMeasuredMetricConfig(selectedYAxisMetric) !== undefined; const legendT = SCATTER_STRINGS[locale]; + // Comparison series (`i_pcompare`) switched off from the legend. Chart-local, + // like Optimal Only's point set: the URL carries the comparison, not which + // of its rows a reader hid while looking. + const [hiddenPowerVariants, setHiddenPowerVariants] = useState>( + () => new Set(), + ); const ephemeralUrlState = useEphemeralUrlState(); const costLimit = chartDefinition.y_cost_limit ?? 0; const latencyLimit = chartDefinition.y_latency_limit ?? 0; @@ -801,7 +885,7 @@ const ScatterGraph = React.memo( () => data.reduce( (acc, point) => { - const key = `${point.hwKey}_${point.precision}`; + const key = scatterSeriesKey(point); if (!acc[key]) acc[key] = []; acc[key].push(point); return acc; @@ -938,7 +1022,7 @@ const ScatterGraph = React.memo( } const buckets = new Map(); const getBucket = (point: InferenceData) => { - const key = `${point.hwKey}|${point.precision}|${point.date}`; + const key = `${scatterSeriesKey(point)}|${point.date}`; let bucket = buckets.get(key); if (!bucket) { bucket = { @@ -987,7 +1071,7 @@ const ScatterGraph = React.memo( const buckets = new Map(); const getBucket = (point: InferenceData) => { const runIndex = overlayRunIndex(point.run_url ?? null, runIndexByUrl); - const key = `${point.hwKey}|${point.precision}|${point.date}|run${runIndex}`; + const key = `${scatterSeriesKey(point)}|${point.date}|run${runIndex}`; let bucket = buckets.get(key); if (!bucket) { bucket = { @@ -1092,6 +1176,8 @@ const ScatterGraph = React.memo( interface Entry { hwKey: string; runIndex: number; + /** Comparison variant id for boundary / role clones, null for the run's base series. */ + variant: string | null; points: InferenceData[]; } if (processedOverlayData.length === 0) return {} as Record; @@ -1100,8 +1186,15 @@ const ScatterGraph = React.memo( const grouped = processedOverlayData.reduce( (acc, p) => { const runIndex = overlayRunIndex(p.run_url ?? null, runIndexByUrl); - const key = `${p.hwKey}_${p.precision}_run${runIndex}`; - if (!acc[key]) acc[key] = { hwKey: String(p.hwKey), runIndex, points: [] }; + const key = `${scatterSeriesKey(p)}_run${runIndex}`; + if (!acc[key]) { + acc[key] = { + hwKey: String(p.hwKey), + runIndex, + variant: p.powerVariant?.id ?? null, + points: [], + }; + } acc[key].points.push(p); return acc; }, @@ -1180,9 +1273,11 @@ const ScatterGraph = React.memo( // its X marker sitting on the dashed roofline and read as a pareto point. const isOverlayPointVisible = useCallback( (d: InferenceData) => + !hiddenPowerVariants.has(powerVariantId(d.powerVariant)) && (!hideNonOptimal || overlayOptimalPoints.has(d)) && (!showPowerEnvelope || showAllMeasurements || overlayEnvelopePoints.has(d)), [ + hiddenPowerVariants, hideNonOptimal, overlayOptimalPoints, showPowerEnvelope, @@ -1216,6 +1311,45 @@ const ScatterGraph = React.memo( ); const { data: persistedLogAvailability } = useLogAvailability(persistedPointIds); const [fixedLogPointId, setFixedLogPointId] = useState(null); + const [powerTelemetryPoint, setPowerTelemetryPoint] = useState(null); + + // "View power trace" on a pinned tooltip (official or overlay point): the + // same-tab click stays in-page — remember which trace to emphasise, switch + // the metric to the Timeline display, and let the anchor's href keep + // serving open-in-new-tab. Listeners are attached per pin because the + // tooltip HTML is replaced on every pin. + const attachPowerTraceAction = useCallback( + (tooltipEl: HTMLElement, d: InferenceData, overlay: boolean) => { + const action = tooltipEl.querySelector('[data-action="view-power-trace"]'); + const traceKey = traceKeyForPoint(d); + if (!action || !traceKey) return; + action.addEventListener('click', (actionEvent) => { + actionEvent.stopPropagation(); + // Modifier / auxiliary clicks keep the anchor's own behaviour: the + // href opens this chart's timeline in a new tab or window. + const mouse = actionEvent as MouseEvent; + if ( + mouse.button !== 0 || + mouse.metaKey || + mouse.ctrlKey || + mouse.shiftKey || + mouse.altKey + ) { + return; + } + actionEvent.preventDefault(); + requestPowerTraceFocus(traceKey); + chartRef.current?.dismissTooltip(); + setSelectedYAxisMetric(POWER_TIMELINE_METRIC_KEY); + track('inference_power_trace_opened', { + hwKey: String(d.hwKey), + conc: d.conc, + overlay, + }); + }); + }, + [setSelectedYAxisMetric], + ); // --- Legend points table (per-series drill-down opened from the legend) --- const [pointsTableTarget, setPointsTableTarget] = useState(null); @@ -1251,6 +1385,7 @@ const ScatterGraph = React.memo( const pts = pointsData.filter( (p) => p.hwKey === hwKey && + !p.powerVariant && selectedPrecisions.includes(p.precision) && (!hideNonOptimal || optimalPointKeys.has(optimalPointKey(p))), ); @@ -1266,6 +1401,7 @@ const ScatterGraph = React.memo( const pts = processedOverlayData.filter( (p) => overlayRunIndex(p.run_url ?? null, runIndexByUrl) === runIndex && + !p.powerVariant && activeOverlayHwTypes.has(p.hwKey as string) && (!hideNonOptimal || overlayOptimalPoints.has(p)), ); @@ -1486,6 +1622,7 @@ const ScatterGraph = React.memo( (d: InferenceData) => effectiveActiveHwTypes.has(d.hwKey as string) && selectedPrecisions.includes(d.precision) && + !hiddenPowerVariants.has(powerVariantId(d.powerVariant)) && (!hideNonOptimal || optimalPointKeys.has(optimalPointKey(d))) && (!showPowerEnvelope || showAllMeasurements || @@ -1493,6 +1630,7 @@ const ScatterGraph = React.memo( [ effectiveActiveHwTypes, selectedPrecisions, + hiddenPowerVariants, hideNonOptimal, optimalPointKeys, showPowerEnvelope, @@ -1669,12 +1807,69 @@ const ScatterGraph = React.memo( getCssColor, ]); + // The comparison in effect and the base series' identity under it. The + // base is the selected metric's own series; deriving it from which variant + // no official point carries breaks when only an overlay carries the + // comparison, and line labels need the same answer as the legend rows. + const powerCompareMode = useMemo(() => { + const official = inferPowerCompare(pointsData); + return official === 'none' ? inferPowerCompare(processedOverlayData) : official; + }, [pointsData, processedOverlayData]); + const powerCompareBaseId = useMemo( + () => powerVariantId(powerCompareBase(selectedYAxisMetric, powerCompareMode)), + [selectedYAxisMetric, powerCompareMode], + ); + + // One legend row per comparison series present (base first). Rows toggle + // chart-local visibility and hover-highlight that series across hardware. + const powerVariantLegendItems = useMemo(() => { + const allPoints = [...pointsData, ...processedOverlayData]; + const variants = powerVariantsInData(allPoints, selectedYAxisMetric); + const baseId = powerCompareBaseId; + return variants.map((variant) => { + const id = powerVariantId(variant); + const legendId = `${POWER_VARIANT_LEGEND_PREFIX}${id}`; + const isBase = id === baseId; + return { + name: legendId, + hw: legendId, + label: powerVariantLabel(variant, locale), + color: 'var(--foreground)', + // The base series is solid, like its points; siblings carry their dash. + lineDasharray: isBase ? '1 0' : powerVariantDash(variant) || '1 0', + isActive: !hiddenPowerVariants.has(isBase ? '' : id), + isRemovable: false, + onClick: () => { + const key = isBase ? '' : id; + setHiddenPowerVariants((prev) => { + const next = new Set(prev); + if (next.has(key)) next.delete(key); + else next.add(key); + return next; + }); + track('inference_power_compare_series_toggled', { + series: id, + visible: hiddenPowerVariants.has(key), + }); + }, + }; + }); + }, [ + pointsData, + processedOverlayData, + selectedYAxisMetric, + powerCompareBaseId, + locale, + hiddenPowerVariants, + ]); + const powerTierCounts = useMemo(() => { - const officialTotal = pointsData.filter((point) => - selectedPrecisions.includes(point.precision), + // Comparison clones re-plot the same measurements; count each once. + const officialTotal = pointsData.filter( + (point) => !point.powerVariant && selectedPrecisions.includes(point.precision), ); - const overlayTotal = processedOverlayData.filter((point) => - selectedPrecisions.includes(point.precision), + const overlayTotal = processedOverlayData.filter( + (point) => !point.powerVariant && selectedPrecisions.includes(point.precision), ); const officialVisible = officialTotal.filter(isPointVisible); const overlayVisible = overlayTotal.filter( @@ -1699,9 +1894,13 @@ const ScatterGraph = React.memo( const hw = el.dataset.hwKey; const prec = el.dataset.precision; if (hw === null || hw === undefined || prec === null || prec === undefined) return false; - return effectiveActiveHwTypes.has(hw) && selectedPrecisions.includes(prec); + return ( + effectiveActiveHwTypes.has(hw) && + selectedPrecisions.includes(prec) && + !hiddenPowerVariants.has(el.dataset.powerVariant ?? '') + ); }, - [effectiveActiveHwTypes, selectedPrecisions], + [effectiveActiveHwTypes, selectedPrecisions, hiddenPowerVariants], ); // --- Interaction state ref --- @@ -1718,6 +1917,7 @@ const ScatterGraph = React.memo( isPointVisible, isOverlayPointVisible, effectiveActiveHwTypes, + hiddenPowerVariants, selectedPrecisions, activeOverlayHwTypes, getCssColor, @@ -1725,11 +1925,13 @@ const ScatterGraph = React.memo( knownIssueAnnotations, traceAvailability, logAvailability: persistedLogAvailability, + showPowerTelemetry, }); interactionRef.current = { isPointVisible, isOverlayPointVisible, effectiveActiveHwTypes, + hiddenPowerVariants, selectedPrecisions, activeOverlayHwTypes, getCssColor, @@ -1737,6 +1939,7 @@ const ScatterGraph = React.memo( knownIssueAnnotations, traceAvailability, logAvailability: persistedLogAvailability, + showPowerTelemetry, }; // --- Perf ruler (opt-in: click two curves, drag the ruler to any iso-x) --- @@ -1748,17 +1951,39 @@ const ScatterGraph = React.memo( // the curves' rendered paths at the iso-x — neither end needs to be a // data point. Multiple rulers accumulate (capped in the pure module); // completing one immediately allows starting the next. - const [preferPerfRulerMode, setPerfRulerMode] = useState(false); + // + // The primary chart's rulers live in the InferenceProvider store so they + // ride along in share links (`i_rulers`) and survive a remount (table + // view toggle). Every other instance — the replay chart, which draws the + // same curve classes, and harnesses mounted without the provider — keeps + // component-local state. Both paths share one `[state, setState]` pair + // below, so the reducers, refs, and draw passes are path-agnostic. + const perfRulerStore = usePerfRulerStore(); + const persistedRulers = perfRulerStore?.chartId === chartId ? perfRulerStore : undefined; + // Rulers only render while the mode is on (and the mode-off effect below + // clears them), so restored share-link rulers — pending or already + // committed by a previous mount — switch the mode on for this instance. + const [preferPerfRulerMode, setPerfRulerMode] = useState( + () => + persistedRulers !== undefined && + (persistedRulers.pending !== null || persistedRulers.state.rulers.length > 0), + ); const perfRulerMode = preferPerfRulerMode && (!showPowerEnvelope || isMeasuredPowerAxis); - const [perfRulerState, setPerfRulerState] = useState(EMPTY_PERF_RULER_STATE); + const [localPerfRulerState, setLocalPerfRulerState] = + useState(EMPTY_PERF_RULER_STATE); + const perfRulerState = persistedRulers ? persistedRulers.state : localPerfRulerState; + const setPerfRulerState = persistedRulers ? persistedRulers.setState : setLocalPerfRulerState; // Changing the x- or y-axis metric (including the x percentile, which // `x_scale_field` encodes) clears every ruler: the curves are redrawn // in different units, so a ruler that persisted would measure a ratio // the user never placed. Runs before the draw pass so no stale ruler - // ever paints over the new curves. + // ever paints over the new curves. Render-time adjustment is only legal + // for this component's own state, so the hook targets the local state; + // the store applies the same reset to persisted rulers inside the + // provider (see usePerfRulerStoreValue). usePerfRulerAxisReset( perfRulerAxisMetricKey(chartDefinition.x_scale_field, selectedYAxisMetric), - setPerfRulerState, + setLocalPerfRulerState, ); // Draw passes read mode/state through refs so toggling off clears the // rulers in the same pre-paint layout pass — lines/labels must never @@ -1824,6 +2049,51 @@ const ScatterGraph = React.memo( [], ); + // Share-link rulers commit only once BOTH curve paths are in the DOM — + // otherwise the prune pass would eat them before their data (i_gpus, + // comparison dates, overlay runs) has arrived. Hidden curves (opacity 0) + // count as present, like for prune. The iso-x is clamped to the pair's + // overlap through the drawn paths, so a rounded or since-shifted iso-x + // still renders; a pair with disjoint spans can never be measured on + // these axes and is dropped. This runs from the draw pass rather than a + // React effect: the chart first draws in a D3Chart-local re-render + // (dimensions are measured after mount), which re-renders nothing here, + // so an effect keyed on our props could miss the first draw and leave + // resolvable rulers pending for the rest of the session. The store is + // read through a ref for the same reason the draw passes read the ruler + // state through refs. Nothing commits while the mode is off (forced off + // by the power envelope, or switched off by the user) — the mode-off + // effect discards pending rulers, and the analytics event must not + // report a restore nobody saw. Draw passes can repeat before React has + // applied a commit, so the pending list handed over is remembered by + // identity and skipped until the store replaces it. + const persistedRulersRef = useRef(persistedRulers); + persistedRulersRef.current = persistedRulers; + const committedPendingRef = useRef(null); + const commitPendingPerfRulers = useCallback( + (zoomGroup: d3.Selection) => { + const store = persistedRulersRef.current; + const pending = store?.pending ?? null; + if (!store || !pending || !perfRulerModeRef.current) return; + if (committedPendingRef.current === pending) return; + const curveExists = (cls: string) => !zoomGroup.select(`.${CSS.escape(cls)}`).empty(); + const resolved: PerfRulerMeasurement[] = []; + const remaining: PerfRulerMeasurement[] = []; + for (const ruler of pending) { + if (!curveExists(ruler.curveA) || !curveExists(ruler.curveB)) { + remaining.push(ruler); + continue; + } + const isoX = clampPerfRulerIsoXToOverlap(ruler.curveA, ruler.curveB, ruler.isoX); + if (isoX !== null) resolved.push({ ...ruler, isoX }); + } + if (remaining.length === pending.length) return; + committedPendingRef.current = pending; + store.commitPending(resolved, remaining.length > 0 ? remaining : null); + }, + [clampPerfRulerIsoXToOverlap], + ); + // Curve click (widened hit strokes): iso-x is the click's x pixel // through the CURRENT rendered x scale, stored in data space. const handlePerfRulerCurveClick = useCallback( @@ -1852,7 +2122,7 @@ const ScatterGraph = React.memo( (point: InferenceData, source: 'official' | 'overlay') => { const ctx = perfRulerDrawCtxRef.current; if (!ctx) return; - const series = `${String(point.hwKey)}_${point.precision}`; + const series = scatterSeriesKey(point); const base = source === 'overlay' ? `overlay-roofline-${series}_run${overlayRunIndex(point.run_url ?? null, runIndexByUrl)}` @@ -1882,8 +2152,12 @@ const ScatterGraph = React.memo( // the switch handler also clears synchronously, this covers // programmatic mode changes). `clearPerfRulers` bails out with the same // reference when there is nothing to clear. + // Share-link rulers still waiting for their curves go too — the user + // switched the tool off, so nothing should surface later. useEffect(() => { - if (!perfRulerMode) setPerfRulerState(clearPerfRulers); + if (perfRulerMode) return; + setPerfRulerState(clearPerfRulers); + persistedRulers?.discardPending(); }, [perfRulerMode]); // Invisible widened hit strokes over every rendered roofline path @@ -2033,6 +2307,7 @@ const ScatterGraph = React.memo( ) => { perfRulerDrawCtxRef.current = { zoomGroup, xScale, yScale, width, height }; syncPerfRulerHitPaths(zoomGroup); + commitPendingPerfRulers(zoomGroup); const state = perfRulerStateRef.current; const entries: PerfRulerRenderEntry[] = []; if (perfRulerModeRef.current && state.rulers.length > 0) { @@ -2086,7 +2361,7 @@ const ScatterGraph = React.memo( ); if (!dragHandles.empty()) dragHandles.call(perfRulerDrag); }, - [syncPerfRulerHitPaths, perfRulerDrag], + [syncPerfRulerHitPaths, commitPendingPerfRulers, perfRulerDrag], ); drawPerfRulerRef.current = drawPerfRuler; @@ -2120,16 +2395,29 @@ const ScatterGraph = React.memo( const svg = chartRef.current?.getSvgElement?.(); if (!svg) return; const root = d3.select(svg); + // A comparison-series legend row highlights that boundary / role + // across every hardware instead of one hardware across series. + const variantId = hwKey.startsWith(POWER_VARIANT_LEGEND_PREFIX) + ? hwKey.slice(POWER_VARIANT_LEGEND_PREFIX.length) + : null; + const matchesPoint = (d: InferenceData) => + variantId === null + ? String(d.hwKey) === hwKey + : powerVariantId(d.powerVariant) === variantId; root .selectAll('.dot-group') .style('opacity', (d) => - isPointVisible(d) ? (String(d.hwKey) === hwKey ? 1 : 0.15) : 0, + isPointVisible(d) ? (matchesPoint(d) ? pointOpacityForVariant(d) : 0.15) : 0, ); root .selectAll('.roofline-path, .official-overflow-continuation') .style('opacity', function () { if (!isRooflineVisible(this)) return 0; - return this.dataset.hwKey === hwKey ? null : '0.15'; + const matches = + variantId === null + ? this.dataset.hwKey === hwKey + : (this.dataset.powerVariant ?? '') === variantId; + return matches ? null : '0.15'; }); root .selectAll('.parallelism-label, .line-label') @@ -2146,7 +2434,7 @@ const ScatterGraph = React.memo( const root = d3.select(svg); root .selectAll('.dot-group') - .style('opacity', (d) => (isPointVisible(d) ? 1 : 0)); + .style('opacity', (d) => (isPointVisible(d) ? pointOpacityForVariant(d) : 0)); root .selectAll('.roofline-path, .official-overflow-continuation') .style('opacity', function () { @@ -2159,9 +2447,16 @@ const ScatterGraph = React.memo( (this as SVGGElement).dataset, effectiveActiveHwTypes, selectedPrecisions, + activeOverlayHwTypes, ); }); - }, [isPointVisible, isRooflineVisible, effectiveActiveHwTypes, selectedPrecisions]); + }, [ + isPointVisible, + isRooflineVisible, + effectiveActiveHwTypes, + selectedPrecisions, + activeOverlayHwTypes, + ]); // --- Zoom config --- const eventPrefix = chartDefinition.chartType === 'e2e' ? 'latency' : 'interactivity'; @@ -2230,6 +2525,7 @@ const ScatterGraph = React.memo( yLabel, selectedYAxisMetric, hardwareConfig, + showPowerTelemetry: interactionRef.current.showPowerTelemetry, runUrl: d.run_url ? updateRepoUrl(d.run_url) : undefined, hasTrace: d.benchmark_type === 'agentic_traces' && isPersistedBenchmarkId(d.id) @@ -2291,6 +2587,15 @@ const ScatterGraph = React.memo( }); }); } + const powerBtn = tooltipEl.querySelector('[data-action="view-power-telemetry"]'); + if (powerBtn && isPersistedBenchmarkId(d.id)) { + powerBtn.addEventListener('click', (event) => { + event.stopPropagation(); + setPowerTelemetryPoint(d); + chartRef.current?.dismissTooltip(); + track('inference_power_telemetry_opened', { id: d.id, hwKey: d.hwKey, conc: d.conc }); + }); + } const logsBtn = tooltipEl.querySelector('[data-action="view-logs"]'); if (logsBtn && typeof d.id === 'number') { logsBtn.addEventListener('click', (btnEvent) => { @@ -2308,10 +2613,12 @@ const ScatterGraph = React.memo( }); }); } + attachPowerTraceAction(tooltipEl, d, false); }, attachToLayer: 1, // scatter layer is index 1 (after rooflines at 0) }), [ + attachPowerTraceAction, xLabel, yLabel, selectedYAxisMetric, @@ -2325,6 +2632,31 @@ const ScatterGraph = React.memo( // --- Layers --- const layers = useMemo((): LayerConfig[] => { + // Line-label identity of one drawn series under a power comparison + // (`i_pcompare`): the base series keeps the hardware key, so pinned + // anchors and hover hooks keep working; a sibling is `::`. + const wattsAxis = metricPlotsWatts(selectedYAxisMetric); + const lineLabelIdentity = (hw: string, points: readonly InferenceData[]) => { + const variant = points[0]?.powerVariant; + const variantId = powerVariantId(variant); + const isBase = !variant || variantId === powerCompareBaseId; + return { variant, variantId, isBase, seriesId: lineLabelSeriesId(hw, variant, isBase) }; + }; + // A sibling's label says which series it is; a flat provisioned boundary + // (TDP, all-in) on a watts axis also states its value. + const lineLabelSuffix = ( + identity: ReturnType, + points: readonly InferenceData[], + ) => + powerLineLabelSuffix(identity.variant, { + isBase: identity.isBase, + locale, + flatWatts: + !identity.isBase && wattsAxis && identity.variant?.kind === 'basis' + ? flatSeriesValue(points.map((point) => point.y)) + : null, + }); + // ── Layer 0: Rooflines + gradient labels (custom) ── const rooflineLayer: CustomLayerConfig = { type: 'custom', @@ -2362,6 +2694,8 @@ const ScatterGraph = React.memo( key: string; hw: string; precision: string; + /** Comparison variant id (`i_pcompare`), '' for the base series. */ + variant: string; points: InferenceData[]; stroke: string; visible: boolean; @@ -2370,10 +2704,11 @@ const ScatterGraph = React.memo( const activeGradientIds = new Set(); Object.entries(displayedRooflines).forEach(([key, pts]) => { - const hw = key.split('_').slice(0, -1).join('_'); - const precision = key.split('_').pop()!; + const { hw, precision, variant } = parseScatterSeriesKey(key); const visible = - ir.effectiveActiveHwTypes.has(hw) && ir.selectedPrecisions.includes(precision); + ir.effectiveActiveHwTypes.has(hw) && + ir.selectedPrecisions.includes(precision) && + !ir.hiddenPowerVariants.has(variant ?? ''); const baseStroke = ir.getCssColor(ir.resolveColor(hw)); // Split into per-date sub-paths so the line never crosses dates. @@ -2417,6 +2752,7 @@ const ScatterGraph = React.memo( key: entryKey, hw, precision, + variant: variant ?? '', points: datePoints, stroke, visible, @@ -2444,9 +2780,12 @@ const ScatterGraph = React.memo( .attr('data-curve-kind', showPowerEnvelope ? 'power-envelope' : 'pareto') .attr('data-hw-key', (d) => d.hw) .attr('data-precision', (d) => d.precision) + .attr('data-power-variant', (d) => d.variant || null) .attr('fill', 'none') .attr('stroke', (d) => d.stroke) .attr('stroke-width', 2.5) + // Comparison siblings share the hardware colour; the dash tells them apart. + .attr('stroke-dasharray', (d) => powerVariantDashById(d.variant) || null) .attr('d', (d) => lineGen(d.points)) .style('transition', 'opacity 150ms ease') .style('opacity', (d) => (d.visible ? 1 : 0)); @@ -2467,10 +2806,11 @@ const ScatterGraph = React.memo( if (showGradientLabels) { Object.entries(allPointLabelsByKey).forEach(([key, pointLabels]) => { if (pointLabels.length < 2) return; - const hw = key.split('_').slice(0, -1).join('_'); - const precision = key.split('_').pop()!; + const { hw, precision, variant } = parseScatterSeriesKey(key); const visible = - ir.effectiveActiveHwTypes.has(hw) && ir.selectedPrecisions.includes(precision); + ir.effectiveActiveHwTypes.has(hw) && + ir.selectedPrecisions.includes(precision) && + !ir.hiddenPowerVariants.has(variant ?? ''); const segments: { label: string; color: string; points: InferenceData[] }[] = []; let cur = { @@ -2568,12 +2908,22 @@ const ScatterGraph = React.memo( // ── Line labels (run name along each roofline) ── let lineLabels: LineLabelPlacement[] = []; + // Comparison variant and label suffix per label key, for the text + // segments and the `data-power-variant` hook on each pill. + const lineLabelMeta = new Map< + string, + { variantId: string; suffix: string; runTag: string } + >(); if (showLineLabels) { const multiPrecision = ir.selectedPrecisions.length > 1; const officialByGroup = new Map(); for (const entry of entries) { if (!entry.visible) continue; - const groupKey = multiPrecision ? entry.key : entry.hw; + // One label per hardware and, under a power comparison, per + // sibling series: the measured line and its boundary / pool + // lines each say which one they are, instead of the longest + // line taking the hardware's only label. + const groupKey = multiPrecision ? entry.key : `${entry.hw}::${entry.variant}`; const previous = officialByGroup.get(groupKey); if (!previous || entry.points.length > previous.points.length) { officialByGroup.set(groupKey, entry); @@ -2582,39 +2932,58 @@ const ScatterGraph = React.memo( const officialSeries: LineLabelSeries[] = [ ...officialByGroup.values(), - ].map((entry) => ({ - key: entry.key, - seriesId: entry.hw, - label: lineLabelText( - entry.hw, - entry.precision, - multiPrecision, - modelLabel, - entry.points, - ), - color: ir.getCssColor(ir.resolveColor(entry.hw)), - points: entry.points, - keepVisibleOnCollision: entry.points.length === 1, - })); + ].map((entry) => { + const identity = lineLabelIdentity(entry.hw, entry.points); + const suffix = lineLabelSuffix(identity, entry.points); + lineLabelMeta.set(entry.key, { variantId: identity.variantId, suffix, runTag: '' }); + return { + key: entry.key, + seriesId: identity.seriesId, + label: `${lineLabelText( + entry.hw, + entry.precision, + multiPrecision, + modelLabel, + entry.points, + )}${suffix}`, + color: ir.getCssColor(ir.resolveColor(entry.hw)), + points: entry.points, + }; + }); + // Runs drawing the same hardware need a run tag on their pills. + const overlayRunsByHw = new Map>(); + for (const group of Object.values(displayedOverlayRooflines)) { + if (!ir.activeOverlayHwTypes.has(group.hwKey)) continue; + if (!overlayRunsByHw.has(group.hwKey)) overlayRunsByHw.set(group.hwKey, new Set()); + overlayRunsByHw.get(group.hwKey)!.add(group.runIndex); + } const overlaySeries: LineLabelSeries[] = Object.entries( displayedOverlayRooflines, ).flatMap(([overlayKey, group]) => { if (!ir.activeOverlayHwTypes.has(group.hwKey)) return []; const info = unofficialRunInfos[group.runIndex]; const precision = group.points[0]?.precision ?? ''; - const runLabel = info - ? getInferenceRunLabel(`✕ ${info.branch || `run ${info.id}`}`, group.points) - : ''; + const hardwareLabel = lineLabelText( + group.hwKey, + precision, + multiPrecision, + modelLabel, + group.points, + ); + const sharesHardware = (overlayRunsByHw.get(group.hwKey)?.size ?? 0) > 1; + const runTag = info && sharesHardware ? overlayRunTag(info) : ''; const label = info - ? multiPrecision - ? `${runLabel} ${getPrecisionLabel(precision as Precision)}` - : runLabel - : lineLabelText(group.hwKey, precision, multiPrecision, modelLabel, group.points); + ? getOverlayLineLabel(hardwareLabel, info, sharesHardware) + : hardwareLabel; + const identity = lineLabelIdentity(group.hwKey, group.points); + const suffix = lineLabelSuffix(identity, group.points); + const key = `overlay-${overlayKey}`; + lineLabelMeta.set(key, { variantId: identity.variantId, suffix, runTag }); return [ { - key: `overlay-${overlayKey}`, - seriesId: group.hwKey, - label, + key, + seriesId: identity.seriesId, + label: `${label}${suffix}`, color: overlayRunColor(group.runIndex), points: group.points, }, @@ -2637,16 +3006,19 @@ const ScatterGraph = React.memo( const labeledKeys = new Set(lineLabels.map((label) => label.key)); for (const entry of entries) { if (labeledKeys.has(entry.key)) continue; + const identity = lineLabelIdentity(entry.hw, entry.points); + const suffix = lineLabelSuffix(identity, entry.points); + lineLabelMeta.set(entry.key, { variantId: identity.variantId, suffix, runTag: '' }); lineLabels.push({ key: entry.key, - seriesId: entry.hw, - label: lineLabelText( + seriesId: identity.seriesId, + label: `${lineLabelText( entry.hw, entry.precision, multiPrecision, modelLabel, entry.points, - ), + )}${suffix}`, color: ir.getCssColor(ir.resolveColor(entry.hw)), x: xScale(entry.points[0].x), y: yScale(entry.points[0].y), @@ -2663,20 +3035,35 @@ const ScatterGraph = React.memo( } renderLineLabels(zoomGroup, lineLabels, { - seriesAttribute: 'data-hw-key', - iconFor: (label) => getLineLabelVendorIcon(label.seriesId), + seriesAttribute: 'data-series-id', + iconFor: (label) => getLineLabelVendorIcon(lineLabelHardwareKey(label.seriesId)), configureGroup: (labelGroup, label) => { labelGroup .attr('data-visible', label.visible ? '1' : '0') + // Legend hover and filter sync key labels by hardware alone; + // the variant names the comparison sibling ('' for the base). + .attr('data-hw-key', lineLabelHardwareKey(label.seriesId)) + .attr('data-power-variant', lineLabelMeta.get(label.key)?.variantId ?? '') .select('.ll-bg') .attr('opacity', 0.95); }, configureText: (text, label) => { - const config = getHardwareConfig(label.seriesId, modelLabel); + const config = getHardwareConfig(lineLabelHardwareKey(label.seriesId), modelLabel); + // Parse the hardware part without the variant suffix, which gets + // its own segment so the engine is still matched at the end. + const meta = lineLabelMeta.get(label.key); + const suffix = meta?.suffix ?? ''; + const runTag = meta?.runTag ?? ''; + let coreLabel = suffix ? label.label.slice(0, -suffix.length) : label.label; + if (runTag) coreLabel = coreLabel.slice(0, -runTag.length); + // Overlay pills lead with the run marker; the hardware behind it is + // parsed like an official pill so the GPU name stays bold. + const marker = coreLabel.startsWith(OVERLAY_LABEL_MARKER) ? OVERLAY_LABEL_MARKER : ''; + coreLabel = coreLabel.slice(marker.length); const hardwareLabel = getDisplayLabel(config); const isHardwareLabel = - label.label === hardwareLabel || label.label.startsWith(`${config.label} `); - const remainingLabel = isHardwareLabel ? label.label.slice(config.label.length) : ''; + coreLabel === hardwareLabel || coreLabel.startsWith(`${config.label} `); + const remainingLabel = isHardwareLabel ? coreLabel.slice(config.label.length) : ''; // Use this curve's resolved suffix, not the generic hwKey label: // official and overlay curves can share a key but differ by run. const engineLabel = @@ -2686,8 +3073,18 @@ const ScatterGraph = React.memo( engineLabel && remainingLabel.endsWith(engineLabel) ? remainingLabel.slice(0, -engineLabel.length) : remainingLabel; + const markerSegments = marker + ? [{ className: 'll-marker', text: marker, fill: 'white', weight: '600' }] + : []; + const runSegments = runTag + ? [{ className: 'll-run', text: runTag, fill: '#d1d5db', weight: '400' }] + : []; + const variantSegments = suffix + ? [{ className: 'll-variant', text: suffix, fill: 'white', weight: '500' }] + : []; const segments = isHardwareLabel ? [ + ...markerSegments, { className: 'll-gpu', text: config.label, fill: 'white', weight: '700' }, ...(precisionLabel ? [ @@ -2709,14 +3106,19 @@ const ScatterGraph = React.memo( }, ] : []), + ...runSegments, + ...variantSegments, ] : [ + ...markerSegments, { className: 'll-plain', - text: label.label, + text: coreLabel, fill: 'white', weight: '600', }, + ...runSegments, + ...variantSegments, ]; text .selectAll('tspan') @@ -2725,7 +3127,24 @@ const ScatterGraph = React.memo( .attr('class', (segment) => segment.className) .attr('fill', (segment) => segment.fill) .attr('font-weight', (segment) => segment.weight) + .attr('x', null) + .attr('dy', null) .text((segment) => segment.text); + // Keep the framework and role visible when a pill is wider than + // the mobile plot, without shrinking its text or dropping fields. + const textX = Number(text.attr('x') ?? 0); + const maxLineWidth = ctx.width - textX - 10; + let lineWidth = 0; + text.selectAll('tspan').each(function () { + const width = this.getComputedTextLength(); + if (lineWidth > 0 && lineWidth + width > maxLineWidth) { + d3.select(this) + .attr('x', textX) + .attr('dy', CHART_TYPE.lineLabel + 3); + lineWidth = 0; + } + lineWidth += width; + }); }, }); // Labels can be joined independently of the Pareto display pass. @@ -2826,11 +3245,11 @@ const ScatterGraph = React.memo( { key: string; seriesId: string; points: InferenceData[] } >(); for (const [key, points] of Object.entries(displayedRooflines)) { - const hardware = key.split('_').slice(0, -1).join('_'); - const precision = key.split('_').pop()!; + const { hw: hardware, precision, variant } = parseScatterSeriesKey(key); if ( !ir.effectiveActiveHwTypes.has(hardware) || - !ir.selectedPrecisions.includes(precision) + !ir.selectedPrecisions.includes(precision) || + ir.hiddenPowerVariants.has(variant ?? '') ) { continue; } @@ -2838,12 +3257,12 @@ const ScatterGraph = React.memo( const singleDate = pointsByDate.size === 1; for (const [date, datePoints] of pointsByDate) { const entryKey = singleDate ? key : `${key}__${encodeURIComponent(date)}`; - const groupKey = multiPrecision ? entryKey : hardware; + const groupKey = multiPrecision ? entryKey : `${hardware}::${variant ?? ''}`; const previous = bestByGroup.get(groupKey); if (!previous || datePoints.length > previous.points.length) { bestByGroup.set(groupKey, { key: entryKey, - seriesId: hardware, + seriesId: lineLabelIdentity(hardware, datePoints).seriesId, points: datePoints, }); } @@ -2854,7 +3273,6 @@ const ScatterGraph = React.memo( ...entry, label: '', color: '', - keepVisibleOnCollision: entry.points.length === 1, }), ); const overlaySeries: LineLabelSeries[] = Object.entries( @@ -2864,7 +3282,7 @@ const ScatterGraph = React.memo( ? [ { key: `overlay-${overlayKey}`, - seriesId: group.hwKey, + seriesId: lineLabelIdentity(group.hwKey, group.points).seriesId, label: '', color: '', points: group.points, @@ -2898,7 +3316,8 @@ const ScatterGraph = React.memo( interactionRef.current.getCssColor( interactionRef.current.resolveColor(d.hwKey as string), ), - getOpacity: (d) => (interactionRef.current.isPointVisible(d) ? 1 : 0), + getOpacity: (d) => + interactionRef.current.isPointVisible(d) ? pointOpacityForVariant(d) : 0, getPointerEvents: (d) => (interactionRef.current.isPointVisible(d) ? 'auto' : 'none'), hideLabels: !showPointLabels || showGradientLabels, // Concurrency (C=) is appended only when the advanced @@ -2908,6 +3327,7 @@ const ScatterGraph = React.memo( dataAttrs: { 'hw-key': (d) => String(d.hwKey), precision: (d) => d.precision, + 'power-variant': (d) => d.powerVariant?.id ?? '', // Lets the agentic coach mark pick an anchor out of the DOM // without knowing anything about React state. 'benchmark-type': (d) => d.benchmark_type ?? '', @@ -2986,6 +3406,7 @@ const ScatterGraph = React.memo( points: InferenceData[]; stroke: string; runIndex: number; + variant: string | null; } const ovEntries: OvEntry[] = []; Object.entries(displayedOverlayRooflines).forEach(([key, group]) => { @@ -2997,6 +3418,7 @@ const ScatterGraph = React.memo( // Color by run — same palette entry the legend uses, so they match. stroke: overlayRunColor(group.runIndex), runIndex: group.runIndex, + variant: group.variant, }); } }); @@ -3014,9 +3436,21 @@ const ScatterGraph = React.memo( .attr('fill', 'none') .attr('stroke', (d) => d.stroke) .attr('stroke-width', 2) - .attr('stroke-dasharray', (d) => overlayRooflineDasharray(d.runIndex)) + .attr('data-power-variant', (d) => d.variant) + // The run keeps its colour; a comparison sibling takes the + // variant dash so it reads like its official counterpart. + .attr('stroke-dasharray', (d) => + d.variant + ? powerVariantDashById(d.variant) + : overlayRooflineDasharray(d.runIndex), + ) .attr('d', (d) => lineGen(d.points)) - .style('filter', null); + .style('filter', null) + // Comparison rows hidden from the legend (the decoration effect + // keeps this in step with later toggles). + .style('opacity', (d) => + interactionRef.current.hiddenPowerVariants.has(d.variant ?? '') ? 0 : null, + ); // Overlay X-shape points — index-keyed so every point renders const overlayPoints = zoomGroup @@ -3052,7 +3486,7 @@ const ScatterGraph = React.memo( overlayPoints.each(function (d) { const visible = interactionRef.current.isOverlayPointVisible(d); d3.select(this) - .style('opacity', visible ? 1 : 0) + .style('opacity', visible ? pointOpacityForVariant(d) : 0) .style('pointer-events', visible ? 'auto' : 'none'); }); overlayPoints @@ -3136,6 +3570,9 @@ const ScatterGraph = React.memo( y: point.y, overlay: true, }); + // The shared helper has just rendered the pinned content into + // this element and pinned it via `handle`. + attachPowerTraceAction(ctx.tooltipElement, point, true); }, }); }, @@ -3406,10 +3843,12 @@ const ScatterGraph = React.memo( xLabel, yLabel, selectedYAxisMetric, + powerCompareBaseId, isMeasuredEnergyAxis, chartDefinition, locale, drawPerfRuler, + attachPowerTraceAction, ]); // Layers handle for the decoration effect — lets it re-run individual @@ -3488,7 +3927,9 @@ const ScatterGraph = React.memo( zoomGroup.selectAll('.dot-group').each(function (d) { const point = d3.select(this); const visible = ir.isPointVisible(d); - point.style('opacity', visible ? 1 : 0).style('pointer-events', visible ? 'auto' : 'none'); + point + .style('opacity', visible ? pointOpacityForVariant(d) : 0) + .style('pointer-events', visible ? 'auto' : 'none'); const color = (showGradientLabels && gradientColorByPoint.get(d)) || ir.getCssColor(ir.resolveColor(d.hwKey as string)); @@ -3507,9 +3948,20 @@ const ScatterGraph = React.memo( zoomGroup.selectAll('.unofficial-overlay-pt').each(function (d) { const visible = ir.isOverlayPointVisible(d); d3.select(this) - .style('opacity', visible ? 1 : 0) + .style('opacity', visible ? pointOpacityForVariant(d) : 0) .style('pointer-events', visible ? 'auto' : 'none'); }); + // Overlay rooflines are only drawn for active overlay hardware; a + // comparison row hidden from the legend is the one visibility toggle + // they answer to here. + zoomGroup.selectAll('.overlay-roofline-path').each(function () { + const roofline = d3.select(this); + if (ir.hiddenPowerVariants.has(this.dataset.powerVariant ?? '')) { + roofline.style('opacity', 0); + } else { + roofline.style('opacity', null); + } + }); // Rooflines: visibility and solid-stroke recolor as direct writes. Keep // gradient url references intact and never touch animated path geometry. @@ -3519,7 +3971,9 @@ const ScatterGraph = React.memo( if (!hw || !precision) return; const roofline = d3.select(this); const visible = - ir.effectiveActiveHwTypes.has(hw) && ir.selectedPrecisions.includes(precision); + ir.effectiveActiveHwTypes.has(hw) && + ir.selectedPrecisions.includes(precision) && + !ir.hiddenPowerVariants.has(this.dataset.powerVariant ?? ''); roofline.style('opacity', visible ? 1 : 0); const stroke = roofline.attr('stroke'); if (stroke && !stroke.startsWith('url(')) { @@ -3553,6 +4007,7 @@ const ScatterGraph = React.memo( (this as SVGGElement).dataset, ir.effectiveActiveHwTypes, ir.selectedPrecisions, + ir.activeOverlayHwTypes, ); }); }, [ @@ -3711,7 +4166,9 @@ const ScatterGraph = React.memo( // brings a hidden ruler back); curves whose paths left the DOM // entirely are truly gone from the data, so prune each ruler (and the // draft) that references one. `prunePerfRulers` bails out with the - // same reference when nothing changed. + // same reference when nothing changed. Share-link rulers still + // pending are not state yet, so prune cannot touch them; drawPerfRuler + // above committed those whose curves now exist. setPerfRulerState((prev) => prunePerfRulers(prev, (cls) => !display.zoomGroup.select(`.${CSS.escape(cls)}`).empty()), ); @@ -3973,6 +4430,9 @@ const ScatterGraph = React.memo( ) : null, })), + // Comparison series (`i_pcompare`): one dash-swatch row per + // boundary / role, toggling that series across every hardware. + ...powerVariantLegendItems, ]} disableActiveSort={false} isLegendExpanded={isLegendExpanded} @@ -4126,7 +4586,10 @@ const ScatterGraph = React.memo( // the pre-paint decoration effect then removes the rulers // and the curve hit strokes before the next frame (no // lingering lines after toggle-off). - if (!checked) setPerfRulerState(clearPerfRulers); + if (!checked) { + setPerfRulerState(clearPerfRulers); + persistedRulers?.discardPending(); + } track('latency_perf_ruler_toggled', { enabled: checked }); }, }, @@ -4174,6 +4637,7 @@ const ScatterGraph = React.memo( count: perfRulerState.rulers.length, }); setPerfRulerState(clearPerfRulers); + persistedRulers?.discardPending(); }, }, ] @@ -4226,6 +4690,15 @@ const ScatterGraph = React.memo( } /> )} + {powerTelemetryPoint === null ? null : ( + { + if (!open) setPowerTelemetryPoint(null); + }} + /> + )} {fixedLogPointId === null ? null : ( { - const ay = getNestedYValue(a, yPath); - const by = getNestedYValue(b, yPath); + const ay = a.powerVariant ? a.y : getNestedYValue(a, yPath); + const by = b.powerVariant ? b.y : getNestedYValue(b, yPath); return yAscending ? ay - by : by - ay; }); } diff --git a/packages/app/src/components/inference/ui/line-label-layer.test.ts b/packages/app/src/components/inference/ui/line-label-layer.test.ts index 33bb77256..e3eda7273 100644 --- a/packages/app/src/components/inference/ui/line-label-layer.test.ts +++ b/packages/app/src/components/inference/ui/line-label-layer.test.ts @@ -14,17 +14,12 @@ interface Point { y: number; } -const series = ( - key: string, - points: Point[], - keepVisibleOnCollision = false, -): LineLabelSeries => ({ +const series = (key: string, points: Point[]): LineLabelSeries => ({ key, seriesId: key, label: key, color: '#000', points, - keepVisibleOnCollision, }); const identity = (value: number) => value; diff --git a/packages/app/src/components/inference/ui/line-label-layer.ts b/packages/app/src/components/inference/ui/line-label-layer.ts index 6da0a7a23..a3b635dfa 100644 --- a/packages/app/src/components/inference/ui/line-label-layer.ts +++ b/packages/app/src/components/inference/ui/line-label-layer.ts @@ -16,7 +16,6 @@ export interface LineLabelSeries { label: string; color: string; points: readonly TPoint[]; - keepVisibleOnCollision?: boolean; } export interface LineLabelPlacement { @@ -26,6 +25,12 @@ export interface LineLabelPlacement { color: string; x: number; y: number; + /** + * `placeLineLabels` always emits `true`: every series it is given keeps its + * pill, overlapping if it must. The only producer of `false` is the caller's + * de-duplication pass, which keeps a hidden data-join entry for a curve that + * lost the one-label-per-hardware contest (GH #470). + */ visible: boolean; } @@ -146,16 +151,16 @@ interface PillLayoutItem { * its anchor, and both mirrors together. Every candidate is clamped into * `bounds` before the overlap test, so nothing leaves the plot. When every * mirrored candidate collides, nearby rows are tried before the default spot - * is kept: an overlapped label is still - * better than a missing one, and the fallback matches what the anchor pass - * already tolerates for pinned anchors. + * is kept: an overlapped label is still better than a missing one, and the + * fallback matches what the anchor pass already tolerates. * * With no bounds — a chart that clips nothing — the anchor offset is applied * unchanged and the collision pass is skipped, preserving that chart's * existing layout. * * Hidden pills get their default transform and occupy no space, so a label - * that later becomes visible reappears where the anchor pass put it. + * that later becomes visible reappears where the anchor pass put it. Only the + * caller's de-duplication pass hides pills; the anchor pass never does. */ function layoutPills( items: readonly PillLayoutItem[], @@ -319,6 +324,21 @@ function lineCandidates( return candidates; } +/** + * Anchor one pill per series along its line. + * + * Each series tries `ANCHOR_SLOTS` fractions along its own points, rotated by + * its index so converging curves spread out instead of stacking at the + * endpoint. A series that finds a clear slot takes it. A series that finds none + * is deferred and placed afterwards on its least crowded slot, so it never + * steals a clear slot from a series that could have used it. + * + * Every series gets a visible pill. The overlap that survives here is resolved + * by `layoutPills`, which runs later with the pills' real measured boxes; the + * crude nominal box used here is far too small to decide that a label is + * unplaceable — a rendered pill is routinely two to three times + * `collisionWidth`. + */ export function placeLineLabels( series: readonly LineLabelSeries[], xScale: (value: number) => number, @@ -346,6 +366,35 @@ export function placeLineLabels( Math.abs(other.y - y) < collisionHeight && Math.abs(other.x - x) < other.halfW + labelHalfWidth, ); + /** + * Nominal overlap area against the labels already placed. The same crude box + * model as `collides`, scored instead of thresholded, so a slot that clips one + * neighbour is preferred over one that sits on three. + */ + const collisionCost = (x: number, y: number) => + placed.reduce((cost, other) => { + const dx = other.halfW + labelHalfWidth - Math.abs(other.x - x); + const dy = collisionHeight - Math.abs(other.y - y); + return dx > 0 && dy > 0 ? cost + dx * dy : cost; + }, 0); + + const emit = (entry: LineLabelSeries, point: TPoint) => { + const x = xScale(point.x); + const y = yScale(point.y); + placed.push({ x, y, halfW: labelHalfWidth }); + result.push({ + key: entry.key, + seriesId: entry.seriesId, + label: entry.label, + color: entry.color, + x, + y, + visible: true, + }); + }; + + /** Series with no clear slot, deferred to a second pass — see below. */ + const crowded: { entry: LineLabelSeries; candidates: TPoint[] }[] = []; for (const [seriesIndex, entry] of sorted.entries()) { if (entry.points.length === 0) continue; @@ -376,35 +425,37 @@ export function placeLineLabels( const candidate = candidates.find((point) => !collides(xScale(point.x), yScale(point.y))); if (candidate) { - const x = xScale(candidate.x); - const y = yScale(candidate.y); - placed.push({ x, y, halfW: labelHalfWidth }); - result.push({ - key: entry.key, - seriesId: entry.seriesId, - label: entry.label, - color: entry.color, - x, - y, - visible: true, - }); + emit(entry, candidate); continue; } - const fallback = entry.points[0]; - const x = xScale(fallback.x); - const y = yScale(fallback.y); - const visible = entry.keepVisibleOnCollision === true; - if (visible) placed.push({ x, y, halfW: labelHalfWidth }); - result.push({ - key: entry.key, - seriesId: entry.seriesId, - label: entry.label, - color: entry.color, - x, - y, - visible, - }); + // No clear slot. Defer rather than claim one now: a series that is going to + // overlap something must not take a slot a later series could have had to + // itself. + crowded.push({ entry, candidates }); + } + + // Every series keeps a pill. `layoutPills` runs after this with the real + // measured boxes and can still mirror it, shift it a row and clamp it into the + // plot — "an overlapped label is still better than a missing one". Emitting the + // crowded ones last also hands that pass the clean labels first, so the crowded + // ones do the moving. + // + // This is where the chart stopped promising that line labels never overlap + // (#132 introduced the drop as the only way to honour that, #434 restated it). + // The promise was worth less than it cost: a dropped pill is silent, and with + // line labels on, PNG export omits the legend, so the series loses its only + // identifier. An overlapping pill at least announces itself. + for (const { entry, candidates } of crowded) { + emit( + entry, + candidates.reduce((best, point) => + collisionCost(xScale(point.x), yScale(point.y)) < + collisionCost(xScale(best.x), yScale(best.y)) + ? point + : best, + ), + ); } return result; diff --git a/packages/app/src/components/inference/ui/line-label-visibility.test.ts b/packages/app/src/components/inference/ui/line-label-visibility.test.ts index 80e17b049..3b4a075d8 100644 --- a/packages/app/src/components/inference/ui/line-label-visibility.test.ts +++ b/packages/app/src/components/inference/ui/line-label-visibility.test.ts @@ -53,6 +53,26 @@ describe('labelOpacityForActiveState', () => { }); }); +describe('labelOpacityForActiveState with ?unofficialrun= overlays', () => { + const official = new Set(['gb200_dynamo-sglang']); + const precisions = ['fp8']; + + it('hides an overlay label when its overlay hardware row is off, even if the official row is on', () => { + expect( + labelOpacityForActiveState( + { + hwKey: 'gb200_dynamo-sglang', + lineKey: 'overlay-gb200_dynamo-sglang_fp8_run1', + visible: '1', + }, + official, + precisions, + new Set(['gb300_dynamo-sglang']), + ), + ).toBe(0); + }); +}); + describe('labelOpacityForHover', () => { it('lights up the kept label for the hovered hardware', () => { expect(labelOpacityForHover({ hwKey: 'b300_sglang', visible: '1' }, 'b300_sglang')).toBe(1); diff --git a/packages/app/src/components/inference/ui/line-label-visibility.ts b/packages/app/src/components/inference/ui/line-label-visibility.ts index a8b34b810..9dfa8e878 100644 --- a/packages/app/src/components/inference/ui/line-label-visibility.ts +++ b/packages/app/src/components/inference/ui/line-label-visibility.ts @@ -25,6 +25,8 @@ export interface LabelAttrs { /** `data-hw-key` — base hardware key, shared across a hw's curves. */ hwKey?: string; + /** `data-line-key` — `overlay-…` marks an unofficial-run curve's label. */ + lineKey?: string; /** `data-precision` — set on parallelism labels, absent on line labels. */ precision?: string; /** `data-visible` — `'1'`/`'0'`; only line labels set this. */ @@ -50,16 +52,24 @@ export const labelOpacityForHover = (attrs: LabelAttrs, hoveredHwKey: string): 0 * filter-change sync effect. Line labels (no precision) show when their * hardware is active **and** the render kept them; parallelism labels show when * their hardware is active and their precision is selected. + * + * An `?unofficialrun=` overlay curve answers to the overlay legend rows, not + * the official ones: its label follows `activeOverlayHwTypes`, so soloing an + * official hardware no longer hides the overlay pills of every other hardware + * (and hiding an official row keeps its overlay twin labelled). */ export const labelOpacityForActiveState = ( attrs: LabelAttrs, activeHwTypes: ReadonlySet, selectedPrecisions: readonly string[], + activeOverlayHwTypes?: ReadonlySet, ): 0 | 1 => { const { hwKey, precision } = attrs; if (!hwKey) return 0; + const isOverlay = attrs.lineKey?.startsWith('overlay-') ?? false; + const active = isOverlay && activeOverlayHwTypes ? activeOverlayHwTypes : activeHwTypes; if (!precision) { - return activeHwTypes.has(hwKey) && renderKept(attrs) ? 1 : 0; + return active.has(hwKey) && renderKept(attrs) ? 1 : 0; } - return activeHwTypes.has(hwKey) && selectedPrecisions.includes(precision) ? 1 : 0; + return active.has(hwKey) && selectedPrecisions.includes(precision) ? 1 : 0; }; diff --git a/packages/app/src/components/inference/utils.ts b/packages/app/src/components/inference/utils.ts index 5284aac3b..1794cd585 100644 --- a/packages/app/src/components/inference/utils.ts +++ b/packages/app/src/components/inference/utils.ts @@ -8,8 +8,15 @@ import { getGpuSpecs, type TcoBasis } from '@/lib/constants'; import chartDefinitions from '@/components/inference/metric-registry'; import { resolveXAxisField } from '@/components/inference/utils/resolveXAxisField'; import { remapInferencePoint } from '@/lib/chart-utils'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; -import type { ChartDefinition, ClippedInferenceData, InferenceData, YAxisMetricKey } from './types'; +import type { + ChartDefinition, + ClippedInferenceData, + InferenceData, + PowerCompare, + YAxisMetricKey, +} from './types'; import type { XAxisMode } from './hooks/useChartData'; /** @@ -145,6 +152,7 @@ export function processOverlayChartData( selectedXAxisMode?: XAxisMode; restrictToNormalizedFrontier?: boolean; tcoBasis?: TcoBasis; + powerCompare?: PowerCompare; }, ): InferenceData[] { return processOverlayChartDataWithClipping( @@ -171,6 +179,8 @@ export function processOverlayChartDataWithClipping( selectedXAxisMode?: XAxisMode; restrictToNormalizedFrontier?: boolean; tcoBasis?: TcoBasis; + /** Sibling boundary / role series, mirroring the official path in useChartData. */ + powerCompare?: PowerCompare; }, ): ProcessedChartData { const chartDef = (chartDefinitions as ChartDefinition[]).find((d) => d.chartType === chartType); @@ -222,9 +232,13 @@ export function processOverlayChartDataWithClipping( // for the natural axis and for agentic (long TTFTs are normal there). const isTtftX = xAxisField.endsWith('_ttft'); - const processedData = sourceData - .filter((d) => metricKey in d) - .map((d) => remapInferencePoint(d, metricKey, xAxisField)); + const processedData = expandPowerCompareSeries( + sourceData + .filter((d) => metricKey in d) + .map((d) => remapInferencePoint(d, metricKey, xAxisField)), + selectedYAxisMetric, + options?.powerCompare ?? 'none', + ); // The normalized metric is derived from persisted request traces, which an // unofficial overlay does not have. An all-false canonical stamp prevents a diff --git a/packages/app/src/components/inference/utils/best-series-per-sku.ts b/packages/app/src/components/inference/utils/best-series-per-sku.ts index 6799b9631..2a35f0b9c 100644 --- a/packages/app/src/components/inference/utils/best-series-per-sku.ts +++ b/packages/app/src/components/inference/utils/best-series-per-sku.ts @@ -40,7 +40,9 @@ export function bestSeriesPerSku(points: InferenceData[], direction: Direction): const bySku = new Map>(); const featured = new Set(); for (const point of points) { - if (!isFrontierEligible(point) || !Number.isFinite(point.y)) continue; + // Comparison clones re-plot the same configs at another boundary or role; + // the best series per SKU is judged on the selected metric alone. + if (point.powerVariant || !isFrontierEligible(point) || !Number.isFinite(point.y)) continue; const sku = baseSku(point); const key = String(point.hwKey); if (point.framework === 'tilert') featured.add(key); diff --git a/packages/app/src/components/inference/utils/point-identity.ts b/packages/app/src/components/inference/utils/point-identity.ts index 2c975593e..0cc8bbed8 100644 --- a/packages/app/src/components/inference/utils/point-identity.ts +++ b/packages/app/src/components/inference/utils/point-identity.ts @@ -23,9 +23,47 @@ export function scatterPointConfigId(point: InferenceData): string { // Agentic series omit spec decoding from hwKey so one curve can mix methods. // It remains point identity to avoid collapsing overlapping MTP/STP results. key += agenticSpecDecodingKeySuffix(point); + // Comparison clones share every config field with their base point. + if (point.powerVariant) key += `|variant-${point.powerVariant.id}`; return key; } +/** + * Comparison-series suffix inside a scatter series key. Letters, digits and + * dashes only, so the key stays a valid CSS class token (the perf ruler and + * `i_rulers` address rooflines by class) and needs no escaping. + */ +const SERIES_VARIANT_DELIMITER = '-v-'; + +/** + * Identity of one drawn series: hardware key, precision and, on a power + * comparison, the boundary or role variant. Rooflines, frontiers, line labels + * and the perf ruler all key on this string. + */ +export function scatterSeriesKey( + point: Pick, +): string { + const base = `${point.hwKey}_${point.precision}`; + return point.powerVariant ? `${base}${SERIES_VARIANT_DELIMITER}${point.powerVariant.id}` : base; +} + +export interface ScatterSeriesIdentity { + hw: string; + precision: string; + /** Comparison variant id (`gpu-provisioned`, `prefill`, …) or null for the base series. */ + variant: string | null; +} + +/** Inverse of `scatterSeriesKey`; hardware keys may themselves contain underscores. */ +export function parseScatterSeriesKey(key: string): ScatterSeriesIdentity { + const delimiter = key.indexOf(SERIES_VARIANT_DELIMITER); + const core = delimiter === -1 ? key : key.slice(0, delimiter); + const variant = delimiter === -1 ? null : key.slice(delimiter + SERIES_VARIANT_DELIMITER.length); + const parts = core.split('_'); + const precision = parts.pop() ?? ''; + return { hw: parts.join('_'), precision, variant }; +} + /** * Stable D3 join key for an official scatter point. * diff --git a/packages/app/src/components/inference/utils/power-compare.ts b/packages/app/src/components/inference/utils/power-compare.ts new file mode 100644 index 000000000..d0dea0b73 --- /dev/null +++ b/packages/app/src/components/inference/utils/power-compare.ts @@ -0,0 +1,326 @@ +/** + * Comparison series for the gated measured-power charts (`i_pcompare`). + * + * The selected metric stays the chart's base series. A comparison adds sibling + * series drawn from the SAME points — one per other power boundary + * (`boundaries`, PowerX Figures 2/3) or per worker role (`roles`, Figures 6/7) + * — so ScatterGraph can draw them with the hardware's colour and a per-variant + * dash through its ordinary series pipeline (frontier, Optimal Only, tooltip, + * table, CSV, `?unofficialrun=` overlay). A variant point is a clone of its + * base point with `y` remapped and `powerVariant` set; base points are left + * untouched so a chart without comparison is byte-identical to before. + * + * Variants exist only where the metric names a whole-deployment average + * (W per chip) or J per output token, because those are the only quantities + * every boundary and role publishes on a common axis; everywhere else the + * comparison yields no series and the control says so. + */ +import type { Locale } from '@/lib/i18n'; +import { + POWER_BASES, + POWER_BASIS_FIELDS, + POWER_BASIS_LABELS, + type PowerBasis, +} from '@/lib/power-basis'; + +import { getMeasuredMetricConfig } from '../measured-metric-config'; +import type { InferenceData, PowerCompare, PowerRole, PowerVariant } from '../types'; + +export const POWER_COMPARE_MODES = [ + 'none', + 'boundaries', + 'roles', +] as const satisfies readonly PowerCompare[]; + +export function parsePowerCompare(value: string | null | undefined): PowerCompare { + return value === 'boundaries' || value === 'roles' ? value : 'none'; +} + +/** Point fields a comparison series can plot; each is a `{ y, roof }` pair. */ +export type PowerSeriesField = keyof Pick< + InferenceData, + | 'measuredAvgPower' + | 'measuredPrefillAvgPower' + | 'measuredDecodeAvgPower' + | 'measuredJPerOutputToken' + | 'measuredDecodeJPerOutputToken' + | 'reconstructedPrefillJPerOutputToken' + | 'gpuProvisionedWatts' + | 'gpuProvisionedJPerOutputToken' + | 'utilityProvisionedWatts' + | 'utilityProvisionedJPerOutputToken' + | 'utilityModeledWatts' + | 'utilityModeledJPerOutputToken' +>; + +export interface PowerCompareSeries { + variant: PowerVariant; + field: PowerSeriesField; +} + +const ROLES = ['all', 'prefill', 'decode'] as const satisfies readonly PowerRole[]; + +const ROLE_WATT_FIELDS: Record = { + all: 'measuredAvgPower', + prefill: 'measuredPrefillAvgPower', + decode: 'measuredDecodeAvgPower', +}; +// Energy per output token per role. The prefill pool's own figure is per input +// token; `reconstructedPrefillJPerOutputToken` carries it onto the output-token +// axis (utils/role-energy.ts). +const ROLE_ENERGY_FIELDS: Record = { + all: 'measuredJPerOutputToken', + prefill: 'reconstructedPrefillJPerOutputToken', + decode: 'measuredDecodeJPerOutputToken', +}; + +const ROLE_LABELS: Record = { + all: { en: 'All GPUs', zh: '全部 GPU' }, + prefill: { en: 'Prefill GPUs', zh: '预填充 GPU' }, + decode: { en: 'Decode GPUs', zh: '解码 GPU' }, +}; + +/** SVG dash per variant; the base series and `all` stay solid. */ +const VARIANT_DASH: Record = { + 'gpu-measured': '', + 'gpu-provisioned': '8 4', + 'utility-provisioned': '3 3', + 'utility-modeled': '10 3 2 3', + all: '', + prefill: '7 3', + decode: '2 3', +}; + +/** + * The quantity a metric key plots on the comparison's common axis, or null + * when the key is not a whole-deployment average W/chip or J per output token. + */ +function comparableQuantity(metric: string): 'watts' | 'energy' | null { + const config = getMeasuredMetricConfig(metric); + if (!config) return null; + if (config.family === 'power') { + return config.scope === 'all' && config.statistic === 'average' && config.display === 'watts' + ? 'watts' + : null; + } + return config.scope === 'all' && config.denominator === 'output' && config.unit === 'joules' + ? 'energy' + : null; +} + +/** + * The base series' own identity under a comparison — the boundary or role the + * selected metric already plots — or null when the metric admits no comparison + * of that kind. + */ +export function powerCompareBase(metric: string, mode: PowerCompare): PowerVariant | null { + const config = getMeasuredMetricConfig(metric); + if (!config || mode === 'none') return null; + if (mode === 'boundaries') { + return comparableQuantity(metric) ? { kind: 'basis', id: config.basis } : null; + } + if (config.basis !== 'gpu-measured') return null; + if (config.family === 'power') { + return config.statistic === 'average' && config.display === 'watts' + ? { kind: 'role', id: config.scope } + : null; + } + // Role energy compares J per output token; a prefill J per input token axis + // has no decode counterpart. + return config.denominator === 'output' && config.unit === 'joules' && config.scope !== 'prefill' + ? { kind: 'role', id: config.scope } + : null; +} + +/** The sibling series a comparison adds to the selected metric (never the base itself). */ +export function powerCompareVariants(metric: string, mode: PowerCompare): PowerCompareSeries[] { + const base = powerCompareBase(metric, mode); + const config = getMeasuredMetricConfig(metric); + if (!base || !config) return []; + if (base.kind === 'basis') { + const quantity = comparableQuantity(metric); + if (!quantity) return []; + return POWER_BASES.filter((basis) => basis !== base.id).map((basis) => ({ + variant: { kind: 'basis', id: basis }, + field: + basis === 'gpu-measured' + ? quantity === 'watts' + ? 'measuredAvgPower' + : 'measuredJPerOutputToken' + : POWER_BASIS_FIELDS[basis][quantity], + })); + } + const fields = config.family === 'power' ? ROLE_WATT_FIELDS : ROLE_ENERGY_FIELDS; + return ROLES.filter((role) => role !== base.id).map((role) => ({ + variant: { kind: 'role', id: role }, + field: fields[role], + })); +} + +/** Whether choosing `mode` on `metric` draws anything. `none` is always available. */ +export function powerCompareAvailable(metric: string, mode: PowerCompare): boolean { + return mode === 'none' || powerCompareVariants(metric, mode).length > 0; +} + +/** + * Appends one clone per comparison series to `points` (already remapped onto + * the selected metric). A point lacking a variant's field contributes nothing + * to that series — never a 0 — so the availability rules of each boundary and + * role carry through unchanged. + */ +export function expandPowerCompareSeries( + points: readonly InferenceData[], + metric: string, + mode: PowerCompare, +): InferenceData[] { + const series = powerCompareVariants(metric, mode); + if (series.length === 0) return [...points]; + const result: InferenceData[] = [...points]; + for (const { variant, field } of series) { + for (const point of points) { + const value = point[field]; + if (!value || !Number.isFinite(value.y)) continue; + result.push({ ...point, y: value.y, roof: value.roof, powerVariant: variant }); + } + } + return result; +} + +/** Comparison mode implied by the variants present in a rendered point set. */ +export function inferPowerCompare(points: readonly InferenceData[]): PowerCompare { + for (const point of points) { + if (point.powerVariant?.kind === 'basis') return 'boundaries'; + if (point.powerVariant?.kind === 'role') return 'roles'; + } + return 'none'; +} + +/** Distinct variants in draw order: the base first, then siblings in canonical order. */ +export function powerVariantsInData( + points: readonly InferenceData[], + metric: string, +): PowerVariant[] { + const mode = inferPowerCompare(points); + if (mode === 'none') return []; + const present = new Set(points.map((point) => point.powerVariant?.id).filter(Boolean)); + const base = powerCompareBase(metric, mode); + const ordered: PowerVariant[] = + mode === 'boundaries' + ? POWER_BASES.map((id) => ({ kind: 'basis', id })) + : ROLES.map((id) => ({ kind: 'role', id })); + return ordered.filter((variant) => variant.id === base?.id || present.has(variant.id)); +} + +export function powerVariantId(variant: PowerVariant | null | undefined): string { + return variant?.id ?? ''; +} + +export function powerVariantDash(variant: PowerVariant | null | undefined): string { + return variant ? VARIANT_DASH[variant.id] : ''; +} + +export function powerVariantLabel(variant: PowerVariant, locale: Locale): string { + return variant.kind === 'basis' + ? POWER_BASIS_LABELS[variant.id][locale] + : ROLE_LABELS[variant.id][locale]; +} + +/** Short boundary names for in-chart line labels; the legend keeps the full names. */ +const BASIS_SHORT_LABELS: Record = { + 'gpu-measured': { en: 'Measured', zh: '实测' }, + 'gpu-provisioned': { en: 'TDP', zh: 'TDP' }, + 'utility-provisioned': { en: 'All-in', zh: '全站' }, + 'utility-modeled': { en: 'PUE modeled', zh: 'PUE 建模' }, +}; + +/** Suffix text a line label carries for one comparison series. */ +export function powerVariantShortLabel(variant: PowerVariant, locale: Locale): string { + return variant.kind === 'basis' + ? BASIS_SHORT_LABELS[variant.id][locale] + : ROLE_LABELS[variant.id][locale]; +} + +/** `700 W`, `1.37 kW`, `19.2 kW`: kilowatts from 1000 W, at most two decimals, zeros trimmed. */ +export function formatWatts(watts: number): string { + if (Math.abs(watts) >= 1000) { + return `${(watts / 1000).toFixed(2).replace(/\.?0+$/u, '')} kW`; + } + return `${Math.round(watts)} W`; +} + +/** + * The value a series holds at every point, or null when it varies. A + * provisioned boundary (TDP, all-in) is one number per hardware, so its line + * label can state it instead of sending the reader to the axis. Non-finite + * values are ignored; an empty series is null. + */ +export function flatSeriesValue(values: readonly number[], relTolerance = 0.005): number | null { + const finite = values.filter((value) => Number.isFinite(value)); + if (finite.length === 0) return null; + const first = finite[0]; + const tolerance = Math.abs(first) * relTolerance; + return finite.every((value) => Math.abs(value - first) <= tolerance) ? first : null; +} + +export interface PowerLineLabelOptions { + /** The selected metric's own series keeps the plain hardware label. */ + isBase: boolean; + locale: Locale; + /** Shared watts of a flat series (`flatSeriesValue`), appended after the name. */ + flatWatts?: number | null; +} + +const LINE_LABEL_SUFFIX_SEPARATOR = ' · '; + +/** Suffix appended to a comparison sibling's line label; '' for the base series. */ +export function powerLineLabelSuffix( + variant: PowerVariant | null | undefined, + opts: PowerLineLabelOptions, +): string { + if (opts.isBase || !variant) return ''; + const watts = + typeof opts.flatWatts === 'number' && Number.isFinite(opts.flatWatts) + ? ` ${formatWatts(opts.flatWatts)}` + : ''; + return `${LINE_LABEL_SUFFIX_SEPARATOR}${powerVariantShortLabel(variant, opts.locale)}${watts}`; +} + +const LINE_LABEL_SERIES_DELIMITER = '::'; + +/** + * Line-label series id: the hardware key for the base series (existing pinned + * anchors and hover hooks key on it) and `::` for a sibling. + */ +export function lineLabelSeriesId( + hw: string, + variant: PowerVariant | null | undefined, + isBase: boolean, +): string { + return isBase || !variant ? hw : `${hw}${LINE_LABEL_SERIES_DELIMITER}${variant.id}`; +} + +/** Inverse of `lineLabelSeriesId`: the hardware key behind a line-label series id. */ +export function lineLabelHardwareKey(seriesId: string): string { + const index = seriesId.indexOf(LINE_LABEL_SERIES_DELIMITER); + return index === -1 ? seriesId : seriesId.slice(0, index); +} + +/** Whether `metric` plots watts per chip, so a flat boundary's label can state its value. */ +export function metricPlotsWatts(metric: string): boolean { + const config = getMeasuredMetricConfig(metric); + return config?.family === 'power' && config.display === 'watts'; +} + +/** + * Series label for a table or CSV row: the point's variant, or the base + * series' identity when the chart is comparing and this is a base point. + */ +export function powerSeriesLabel( + point: Pick, + metric: string, + mode: PowerCompare, + locale: Locale, +): string { + const variant = point.powerVariant ?? powerCompareBase(metric, mode); + return variant ? powerVariantLabel(variant, locale) : ''; +} diff --git a/packages/app/src/components/inference/utils/powerCurves.test.ts b/packages/app/src/components/inference/utils/powerCurves.test.ts index a71f8bde2..dc0125649 100644 --- a/packages/app/src/components/inference/utils/powerCurves.test.ts +++ b/packages/app/src/components/inference/utils/powerCurves.test.ts @@ -94,8 +94,12 @@ describe('upper power envelope', () => { it('mirrors the boundary for latency and resolves tied coordinates deterministically', () => { const fast = point(1, 1, 350); const middle = point(8, 2, 700); + const plateau = point(16, 3, 700); const slow = point(32, 4, 950); - const samples = [slow, point(16, 3, 700), point(4, 2, 500), middle, { ...middle }, fast]; + // Measured watts: a tie at the running maximum is a repeat marker, so the + // plateau leaves the boundary and Optimal Only can collapse it; a repeated + // X keeps only its first vertex. + const samples = [slow, plateau, point(4, 2, 500), middle, { ...middle }, fast]; expect(upperPowerEnvelope(samples, false)).toEqual([fast, middle, slow]); expect( upperPowerEnvelope( @@ -103,6 +107,8 @@ describe('upper power envelope', () => { true, ).map((p) => p.y), ).toEqual([950, 700, 350]); + // A gauge keeps the plateau: it is part of the outer edge it draws. + expect(upperPowerEnvelope(samples, false, true)).toEqual([fast, middle, plateau, slow]); }); it('uses only finite positive coordinates and preserves singleton boundaries', () => { diff --git a/packages/app/src/components/inference/utils/powerCurves.ts b/packages/app/src/components/inference/utils/powerCurves.ts index 35146640a..13ac7209e 100644 --- a/packages/app/src/components/inference/utils/powerCurves.ts +++ b/packages/app/src/components/inference/utils/powerCurves.ts @@ -1,3 +1,4 @@ +import { isPowerBasisConfigKey } from '@/components/inference/metric-registry'; import type { InferenceData } from '@/components/inference/types'; import { isFrontierEligible, @@ -15,14 +16,40 @@ const POWER_CURVE_METRICS: ReadonlySet = new Set([ 'y_measuredDecodeAvgPower', 'y_measuredPowerPercentTdp', 'y_modeledChassisPowerPerGpu', + // Provisioned / modelled boundary gauges: same upper-envelope curve as measured watts. + 'y_gpuProvisionedWatts', + 'y_utilityProvisionedWatts', + 'y_utilityModeledWatts', ]); export function isPowerCurveMetric(metric: string): boolean { return POWER_CURVE_METRICS.has(metric); } +/** + * Power gauges whose curve is always the upper envelope and whose Optimal Only + * toggle only hides off-envelope markers: measured watts plus the provisioned / + * modelled boundary gauges that sit beside them. A Pareto corner would collapse + * a flat TDP series to one marker. The modelled chassis axis keeps its legacy + * Pareto behaviour. + */ export function isMeasuredPowerCurveMetric(metric: string): boolean { - return isPowerCurveMetric(metric) && metric !== 'y_modeledChassisPowerPerGpu'; + return ( + isPowerCurveMetric(metric) && + (metric !== 'y_modeledChassisPowerPerGpu' || isPowerBasisConfigKey(metric)) + ); +} + +/** + * Whether a drawn series is a provisioned or modelled gauge rather than + * telemetry: a boundary axis, or a boundary comparison clone of one. Gauges + * keep envelope ties (see `upperPowerEnvelope`); measured series, the measured + * boundary clone and role clones stay on the strict envelope. + */ +export function isPowerGaugeSeries(metric: string, sample: InferenceData | undefined): boolean { + const variant = sample?.powerVariant; + if (variant) return variant.kind === 'basis' && variant.id !== 'gpu-measured'; + return isPowerBasisConfigKey(metric); } /** No declared direction means there is no Pareto frontier to draw or filter by. */ @@ -45,14 +72,23 @@ export function chartFrontier( export function upperPowerEnvelope( points: readonly InferenceData[], maximizeX: boolean, + keepTies = false, ): InferenceData[] { const sorted = points .filter((point) => isFrontierEligible(point) && Number.isFinite(point.y) && point.y > 0) .sort((a, b) => (maximizeX ? b.x - a.x : a.x - b.x) || b.y - a.y); + // Measured telemetry keeps the strict envelope: a tie at the running maximum + // is a repeat marker that Optimal Only collapses. A provisioned or modelled + // gauge (`keepTies`) is flat by construction, so its ties stay on the + // boundary and the series draws across its tested x-range instead of one + // marker. Repeated X keeps only its first vertex so the smoothing never + // backtracks. let maxY = -Infinity; + let lastX = Number.NaN; const envelope = sorted.filter((point) => { - if (point.y <= maxY) return false; + if (point.y < maxY || (point.y === maxY && (!keepTies || point.x === lastX))) return false; maxY = point.y; + lastX = point.x; return true; }); return maximizeX ? envelope.toReversed() : envelope; diff --git a/packages/app/src/components/inference/utils/role-energy.test.ts b/packages/app/src/components/inference/utils/role-energy.test.ts new file mode 100644 index 000000000..102703cc0 --- /dev/null +++ b/packages/app/src/components/inference/utils/role-energy.test.ts @@ -0,0 +1,26 @@ +import { describe, expect, it } from 'vitest'; + +import { reconstructedRoleEnergy } from './role-energy'; + +// A disaggregated 8K/1K run: the deployment's energy over the window divided +// by 8× more input than output tokens, so J/out ÷ J/in = 7.9 = the served ratio. +const entry = { + disagg: true, + power_valid: 1, + power_metric_schema_version: 2, + joules_per_input_token: 1, + joules_per_output_token: 7.9, + prefill_joules_per_input_token: 0.25, + decode_joules_per_output_token: 6, +}; + +describe('reconstructedRoleEnergy', () => { + it('expresses prefill energy per output token with the served token ratio and sums the roles', () => { + expect(reconstructedRoleEnergy(entry)).toEqual({ + prefill: 1.975, + decode: 6, + total: 7.975, + prefillShare: (100 * 1.975) / 7.975, + }); + }); +}); diff --git a/packages/app/src/components/inference/utils/role-energy.ts b/packages/app/src/components/inference/utils/role-energy.ts new file mode 100644 index 000000000..89e6b37f2 --- /dev/null +++ b/packages/app/src/components/inference/utils/role-energy.ts @@ -0,0 +1,64 @@ +/** + * Reconstructs how a disaggregated deployment's request energy splits between + * its prefill and decode pools (PowerX Figure 7). + * + * Schema-2 aggregate energy has one numerator: the deployment's energy over the + * validated window is divided by input tokens for `joules_per_input_token` and + * by output tokens for `joules_per_output_token`. Their ratio is therefore the + * input:output token ratio the benchmark actually served. Multiplying the + * prefill pool's J per input token by that ratio expresses the prefill energy + * per output token, on the same axis as the decode pool's J per output token. + * The two add up to the whole request's J per output token, which equals the + * deployment figure whenever the role energies partition the deployment energy. + * + * Nothing is estimated: every input is a same-window telemetry figure, and the + * result is `undefined` whenever one is missing, not validated, or not from a + * disaggregated deployment. + */ +export interface RoleEnergyInput { + disagg?: boolean; + power_valid?: number; + power_metric_schema_version?: number; + joules_per_input_token?: number; + joules_per_output_token?: number; + prefill_joules_per_input_token?: number; + decode_joules_per_output_token?: number; +} + +export interface ReconstructedRoleEnergy { + /** Prefill pool energy per output token (J). */ + prefill: number; + /** Decode pool energy per output token (J). */ + decode: number; + /** Prefill + decode (J per output token). */ + total: number; + /** Prefill share of the reconstructed total, in percent. */ + prefillShare: number; +} + +const positive = (value: unknown): value is number => + typeof value === 'number' && Number.isFinite(value) && value > 0; + +export function reconstructedRoleEnergy( + entry: RoleEnergyInput, +): ReconstructedRoleEnergy | undefined { + if (!entry.disagg || entry.power_valid !== 1 || entry.power_metric_schema_version !== 2) { + return undefined; + } + const input = entry.joules_per_input_token; + const output = entry.joules_per_output_token; + const prefill = entry.prefill_joules_per_input_token; + const decode = entry.decode_joules_per_output_token; + if (!positive(input) || !positive(output) || !positive(prefill) || !positive(decode)) { + return undefined; + } + const prefillPerOutputToken = prefill * (output / input); + const total = prefillPerOutputToken + decode; + if (!positive(prefillPerOutputToken) || !positive(total)) return undefined; + return { + prefill: prefillPerOutputToken, + decode, + total, + prefillShare: (100 * prefillPerOutputToken) / total, + }; +} diff --git a/packages/app/src/components/inference/utils/tooltip-utils.power-trace.test.ts b/packages/app/src/components/inference/utils/tooltip-utils.power-trace.test.ts new file mode 100644 index 000000000..45157d317 --- /dev/null +++ b/packages/app/src/components/inference/utils/tooltip-utils.power-trace.test.ts @@ -0,0 +1,94 @@ +import { afterEach, describe, expect, it, vi } from 'vitest'; + +import type { HardwareConfig, InferenceData, OverlayData } from '@/components/inference/types'; +import { + generateOverlayTooltipContent, + type OverlayTooltipConfig, + type TooltipConfig, +} from '@/components/inference/utils/tooltipUtils'; + +// "View power trace" on pinned tooltips: the deep link from a measured-power +// scatter point to its per-second telemetry on the Timeline display. + +const RUN_URL = 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34716669498'; +const AUDIT_NAME = + 'dsv4_8k1k_fp4_sglang_tp8-pp1-dcp1-pcp1-ep1-dpafalse_disagg-false_spec-none_conc64_b200-host-0123'; +const ACTION = 'data-action="view-power-trace"'; + +const hardwareConfig = { + b200: { + name: 'b200', + label: 'B200', + suffix: '', + gpu: 'B200', + color: 'blue', + power: 1000, + costh: 5, + costr: 1.25, + }, +} as unknown as HardwareConfig; + +function measuredPoint(overrides: Partial = {}): InferenceData { + return { + id: 980001, + date: '2026-09-01', + x: 60, + y: 600, + tp: 8, + conc: 64, + hwKey: 'b200', + precision: 'fp4', + benchmark_type: 'single_turn', + run_url: RUN_URL, + power_audit: { source: `power_validation_${AUDIT_NAME}.json` }, + tpPerGpu: { y: 400, roof: false }, + tpPerMw: { y: 50, roof: false }, + costh: { y: 1, roof: false }, + costr: { y: 1, roof: false }, + costhi: { y: 1, roof: false }, + costri: { y: 1, roof: false }, + ...overrides, + } as InferenceData; +} + +function config(overrides: Partial = {}): TooltipConfig { + return { + data: measuredPoint(), + isPinned: true, + xLabel: 'Interactivity (tok/s/user)', + yLabel: 'Measured Power per Chip (W)', + selectedYAxisMetric: 'y_measuredAvgPower', + hardwareConfig, + ...overrides, + }; +} + +function overlayConfig(overrides: Partial = {}): OverlayTooltipConfig { + return { + ...config({ data: measuredPoint({ id: 0 }) }), + overlayData: { + label: 'powerx-timeline', + hardwareConfig, + data: [], + runUrl: RUN_URL, + } as unknown as OverlayData, + ...overrides, + }; +} + +describe('View power trace tooltip action', () => { + afterEach(() => { + vi.unstubAllGlobals(); + }); + + it('renders on pinned overlay tooltips whose points carry id 0', () => { + const html = generateOverlayTooltipContent(overlayConfig()); + expect(html).toContain(` { + const variant = d.powerVariant; + if (!variant) return ''; + const t = TOOLTIP_STRINGS[locale]; + let html = tooltipLine(t.series, powerVariantLabel(variant, locale)); + if ( + variant.kind === 'role' && + variant.id !== 'all' && + getMeasuredMetricConfig(selectedYAxisMetric)?.family === 'energy' + ) { + const energy = reconstructedRoleEnergy(d); + if (energy) { + const share = variant.id === 'prefill' ? energy.prefillShare : 100 - energy.prefillShare; + html += tooltipLine(t.roleEnergyShare, `${share.toFixed(1)}%`); + } + } + return html; +}; + const totalChipsHTML = (d: InferenceData, selectedYAxisMetric: string, locale: Locale): string => { const t = TOOLTIP_STRINGS[locale]; const { physical, configured } = chipCounts( @@ -228,6 +270,8 @@ const SYSTEM_POWER_STRINGS = { : `${chassis} eight-GPU chassis · ${measured} of ${modeled} GPUs measured, extrapolated to full chassis`, extrapolation: 'Unmeasured chassis GPUs are assumed to run the same workload at the measured per-GPU power; deployment values are the measured GPUs’ share.', + uniformHosts: + 'No per-host telemetry for this multinode deployment; every chassis is modeled at the deployment-mean GPU power.', normalization: 'AC power is divided by all modeled chassis GPUs, including prefill and decode.', boundary: 'Includes GPU chassis CPUs; excludes separate CPU-only frontend/router hosts.', model: 'Power model source', @@ -257,6 +301,7 @@ const SYSTEM_POWER_STRINGS = { : `${chassis} 个八卡机箱 · 实测 ${measured}/${modeled} 张 GPU,按满机箱外推`, extrapolation: '假设机箱内未实测的 GPU 运行相同负载、功耗与实测每卡功耗相同;部署数值为实测 GPU 所占份额。', + uniformHosts: '该多节点部署没有逐主机功耗数据;每个机箱按部署平均每卡功耗建模。', normalization: '交流功耗按所有建模机箱的 GPU 总数分摊,包括 Prefill 与 Decode。', boundary: '计入 GPU 机箱内的 CPU;不计入独立的纯 CPU 前端或路由主机。', model: '功耗模型来源', @@ -303,7 +348,7 @@ const modeledSystemPowerHTML = ( ? ` ${tooltipLine(t.deploymentAc, `${fmt(estimate.deploymentAcWatts)} W`)} ${tooltipLine(`${t.facility} (PUE ${fmt(estimate.pue)})`, `${fmt(estimate.deploymentFacilityWatts)} W`)} -
    ${t.topology(estimate.chassisCount, estimate.gpuCount, estimate.modeledGpuCount)}${estimate.chassisBasis === 'extrapolated' ? `
    ${t.extrapolation}` : ''}
    ${t.assumptions}
    ${t.platformAssumptions}
    ${t.normalization}
    ${t.boundary}
    +
    ${t.topology(estimate.chassisCount, estimate.gpuCount, estimate.modeledGpuCount)}${estimate.chassisBasis === 'extrapolated' ? `
    ${t.extrapolation}` : ''}${estimate.topologyBasis === 'uniform-hosts' ? `
    ${t.uniformHosts}` : ''}
    ${t.assumptions}
    ${t.platformAssumptions}
    ${t.normalization}
    ${t.boundary}
    ${tooltipLine(t.model, `
    ${escapeHtml(estimate.hardware)} · ${escapeHtml(estimate.modelRevision.slice(0, 12))}`)} ${t.sweep} ` @@ -493,39 +538,108 @@ const generateAgenticHTML = (d: InferenceData, locale: Locale): string => { }; const ACTION_STRINGS = { - en: { charts: 'View charts', logs: 'View logs' }, - zh: { charts: '查看图表', logs: '查看日志' }, + en: { + charts: 'View charts', + logs: 'View logs', + powerTelemetry: 'View PowerX', + powerTrace: 'View power trace', + }, + zh: { + charts: '查看图表', + logs: '查看日志', + powerTelemetry: '查看 PowerX', + powerTrace: '查看功耗曲线', + }, } as const; -const pointDetailActionLink = (action: 'view-charts' | 'view-logs', href: string, label: string) => +type TooltipAction = 'view-charts' | 'view-logs' | 'view-power-trace'; + +const pointDetailActionLink = (action: TooltipAction, href: string, label: string) => `${label} →`; -/** Point-detail links rendered only for persisted, pinned official points. */ -const viewActionsHTML = ( - isPinned: boolean, - hasTraceData: boolean, - hasLogData: boolean, - pointId: number | undefined, - benchmarkType: string | undefined, - locale: Locale, -): string => { - const isAgentic = benchmarkType === 'agentic_traces'; - const showCharts = isAgentic && hasTraceData; - if (!isPinned || !isPersistedBenchmarkId(pointId) || (!showCharts && !hasLogData)) return ''; - const prefix = locale === 'zh' ? '/zh' : ''; - const agenticHref = agenticDetailHref(pointId, locale); - const logHref = isAgentic - ? `${agenticHref}${agenticHref.includes('?') ? '&' : '?'}view=logs` - : `${prefix}/inference/logs/${pointId}`; +/** + * Whether a point on the measured-power / energy scatter can jump to its + * per-second telemetry on the Timeline display. Overlay points qualify too: + * the trace is keyed by run id and audit name, not by a persisted row id. + */ +export const showsPowerTraceAction = ( + point: Pick, + selectedYAxisMetric: string, +): boolean => + selectedYAxisMetric !== POWER_TIMELINE_METRIC_KEY && + getMeasuredMetricConfig(selectedYAxisMetric) !== undefined && + traceKeyForPoint(point) !== null; + +/** + * Same-tab click is intercepted by the chart (in-page metric switch); the href + * keeps open-in-new-tab landing on the Timeline display of THIS chart. The + * address bar is stripped of chart state after load, so the share-link store + * is layered over the live location first (`chartStateHref`). + */ +const powerTraceHref = (): string => + typeof window === 'undefined' ? '#' : chartStateHref({ i_metric: POWER_TIMELINE_METRIC_KEY }); + +interface ViewActionsInput { + isPinned: boolean; + hasTraceData: boolean; + hasLogData: boolean; + point: InferenceData; + /** + * Metric the chart currently plots, when that chart can switch to the + * Timeline display in place (the scatter). Omitted by charts that cannot, + * so they never render a "View power trace" link nothing would handle. + */ + powerTraceMetric?: string; + showPowerTelemetry?: boolean; + locale: Locale; +} + +/** + * Point-detail links rendered only on pinned tooltips. "View charts" and + * "View logs" need a persisted row id (overlay points have none); "View power + * trace" needs only the run and audit source, so it works for overlays too. + */ +const viewActionsHTML = ({ + isPinned, + hasTraceData, + hasLogData, + point, + powerTraceMetric, + showPowerTelemetry, + locale, +}: ViewActionsInput): string => { + if (!isPinned) return ''; const t = ACTION_STRINGS[locale]; - const actions = [ - showCharts ? pointDetailActionLink('view-charts', agenticHref, t.charts) : '', - hasLogData ? pointDetailActionLink('view-logs', logHref, t.logs) : '', - ].filter(Boolean); + const actions: string[] = []; + const pointId = point.id; + const isAgentic = point.benchmark_type === 'agentic_traces'; + const showCharts = isAgentic && hasTraceData; + if (isPersistedBenchmarkId(pointId) && (showCharts || hasLogData)) { + const prefix = locale === 'zh' ? '/zh' : ''; + const agenticHref = agenticDetailHref(pointId, locale); + if (showCharts) { + actions.push(pointDetailActionLink('view-charts', agenticHref, t.charts)); + } + if (hasLogData) { + const logHref = isAgentic + ? `${agenticHref}${agenticHref.includes('?') ? '&' : '?'}view=logs` + : `${prefix}/inference/logs/${pointId}`; + actions.push(pointDetailActionLink('view-logs', logHref, t.logs)); + } + } + if (showPowerTelemetry && isPersistedBenchmarkId(pointId)) { + actions.push( + ``, + ); + } + if (powerTraceMetric !== undefined && showsPowerTraceAction(point, powerTraceMetric)) { + actions.push(pointDetailActionLink('view-power-trace', powerTraceHref(), t.powerTrace)); + } + if (actions.length === 0) return ''; return `
    ${actions.join('')}
    `; }; @@ -708,6 +822,7 @@ export const generateTooltipContent = (config: TooltipConfig): string => { : '' } ${powerTierHTML(d, selectedYAxisMetric, locale)} + ${powerVariantHTML(d, selectedYAxisMetric, locale)} ${modeledSystemPowerHTML(d, selectedYAxisMetric, isPinned, locale)} ${totalChipsHTML(d, selectedYAxisMetric, locale)} ${generateParallelismHTML(d, locale)} @@ -718,7 +833,15 @@ export const generateTooltipContent = (config: TooltipConfig): string => { ${generateAgenticHTML(d, locale)} ${generateWorkerPowerHTML(d, isPinned, locale)} ${runLinkHTML(runUrl, locale)} - ${viewActionsHTML(isPinned, Boolean(hasTrace), Boolean(config.hasLog), d.id, d.benchmark_type, locale)} + ${viewActionsHTML({ + isPinned, + hasTraceData: Boolean(hasTrace), + hasLogData: Boolean(config.hasLog), + point: d, + showPowerTelemetry: config.showPowerTelemetry, + powerTraceMetric: selectedYAxisMetric, + locale, + })} `; }; @@ -752,6 +875,7 @@ export const generateOverlayTooltipContent = (config: OverlayTooltipConfig): str ${tooltipLine(xLabel, fmt(d.x))} ${tooltipLine(yLabel, fmt(d.y))} ${powerTierHTML(d, selectedYAxisMetric, locale)} + ${powerVariantHTML(d, selectedYAxisMetric, locale)} ${modeledSystemPowerHTML(d, selectedYAxisMetric, isPinned, locale)} ${totalChipsHTML(d, selectedYAxisMetric, locale)} ${generateParallelismHTML(d, locale)} @@ -761,6 +885,14 @@ export const generateOverlayTooltipContent = (config: OverlayTooltipConfig): str ${powerWithheldHTML(d, locale)} ${generateAgenticHTML(d, locale)} ${generateWorkerPowerHTML(d, isPinned, locale)} + ${viewActionsHTML({ + isPinned, + hasTraceData: false, + hasLogData: false, + point: d, + powerTraceMetric: selectedYAxisMetric, + locale, + })} `; }; @@ -813,6 +945,7 @@ export const generateGPUGraphTooltipContent = (config: TooltipConfig): string => : '' } ${powerTierHTML(d, selectedYAxisMetric, locale)} + ${powerVariantHTML(d, selectedYAxisMetric, locale)} ${modeledSystemPowerHTML(d, selectedYAxisMetric, isPinned, locale)} ${totalChipsHTML(d, selectedYAxisMetric, locale)} ${generateParallelismHTML(d, locale)} @@ -823,7 +956,14 @@ export const generateGPUGraphTooltipContent = (config: TooltipConfig): string => ${generateAgenticHTML(d, locale)} ${generateWorkerPowerHTML(d, isPinned, locale)} ${runLinkHTML(runUrl, locale)} - ${viewActionsHTML(isPinned, Boolean(hasTrace), Boolean(hasLog), d.id, d.benchmark_type, locale)} + ${viewActionsHTML({ + isPinned, + hasTraceData: Boolean(hasTrace), + hasLogData: Boolean(hasLog), + point: d, + showPowerTelemetry: config.showPowerTelemetry, + locale, + })} `; }; diff --git a/packages/app/src/lib/chart-utils.test.ts b/packages/app/src/lib/chart-utils.test.ts index 9d5203b66..a23699602 100644 --- a/packages/app/src/lib/chart-utils.test.ts +++ b/packages/app/src/lib/chart-utils.test.ts @@ -928,6 +928,89 @@ describe('createChartDataPoint energy fields', () => { }); }); +// =========================================================================== +// createChartDataPoint — power-boundary fields (B2 GPU provisioned, B3 utility +// provisioned, B4 utility modeled). Mock specs: tdp 700 W, power 700 "kW". +// =========================================================================== +const boundaryPoint = (e: AggDataEntry) => + createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); + +describe('createChartDataPoint power-boundary fields', () => { + // Eight measured GPUs (2 prefill + 6 decode) on two partially filled chassis: + // the model evaluates 16 GPUs, so only deploymentFacilityWatts ÷ gpuCount yields 877.5 W. + // Every other numerator/denominator pairing gives a different number. + const supportedModel = { + status: 'supported' as const, + hardware: 'h100', + modelRevision: 'test', + modelPath: 'test', + gpuCount: 8, + chassisCount: 2, + modeledGpuCount: 16, + measuredGpuWattsPerGpu: 500, + chassisAcWatts: 10_400, + chassisAcWattsPerGpu: 650, + facilityWatts: 13_520, + deploymentAcWatts: 5400, + deploymentFacilityWatts: 7020, + pue: 1.3, + telemetryBasis: 'validated-v2' as const, + topologyBasis: 'worker-hosts' as const, + chassisBasis: 'extrapolated' as const, + }; + const validated = { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 500, + joules_per_output_token: 10, + modeledSystemPower: supportedModel, + }; + it('emits all six boundary fields for a validated official row', () => { + const p = boundaryPoint( + entry({ output_tput_per_gpu: 400, benchmark_type: 'single_turn', ...validated }), + ); + expect(p.gpuProvisionedWatts).toEqual({ y: 700, roof: false }); + expect(p.gpuProvisionedJPerOutputToken?.y).toBeCloseTo(700 / 400, 10); + expect(p.utilityProvisionedWatts).toEqual({ y: 700_000, roof: false }); + expect(p.utilityProvisionedJPerOutputToken?.y).toBeCloseTo(700_000 / 400, 10); + // B4 W = deployment facility ÷ measured GPUs; J scales B1 by B4 W ÷ B1 W. + expect(p.utilityModeledWatts).toEqual({ y: 877.5, roof: false }); + expect(p.utilityModeledJPerOutputToken?.y).toBeCloseTo((10 * 877.5) / 500, 10); + // Existing measured (B1) fields are untouched by the new boundaries. + expect(p.measuredAvgPower).toEqual({ y: 500, roof: false }); + expect(p.measuredJPerOutputToken).toEqual({ y: 10, roof: false }); + }); + + it('normalizes fixed-sequence disaggregated energy by all GPUs while jOutput stays per decode GPU', () => { + const p = boundaryPoint( + entry({ + output_tput_per_gpu: 400, + disagg: true, + benchmark_type: 'single_turn', + num_prefill_gpu: 4, + num_decode_gpu: 4, + }), + ); + expect(p.gpuProvisionedJPerOutputToken?.y).toBeCloseTo((700 * 8) / (400 * 4), 10); + expect(p.utilityProvisionedJPerOutputToken?.y).toBeCloseTo((700_000 * 8) / (400 * 4), 10); + expect(p.jOutput?.y).toBeCloseTo(700_000 / 400, 10); + expect(p.utilityProvisionedJPerOutputToken?.y).toBeCloseTo(2 * p.jOutput!.y, 10); + + const agentic = boundaryPoint( + entry({ + output_tput_per_gpu: 400, + disagg: true, + benchmark_type: 'agentic_traces', + num_prefill_gpu: 4, + num_decode_gpu: 4, + }), + ); + expect(agentic.gpuProvisionedWatts?.y).toBe(700); + expect(agentic.gpuProvisionedJPerOutputToken).toBeUndefined(); + expect(agentic.utilityProvisionedJPerOutputToken).toBeUndefined(); + }); +}); + // =========================================================================== // createChartDataPoint — measured power / energy fields (from runner telemetry) // =========================================================================== @@ -952,20 +1035,6 @@ describe('createChartDataPoint measured power fields', () => { expect(missing.measuredP75Power).toBeUndefined(); expect(missing.measuredP90Power).toBeUndefined(); }); - it('emits measuredAvgPower when avg_power_w is present on the entry', () => { - const e = entry({ avg_power_w: 685.5 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredAvgPower).toBeDefined(); - expect(point.measuredAvgPower!.y).toBe(685.5); - expect(point.measuredAvgPower!.roof).toBe(false); - }); - - it('emits measuredJPerOutputToken when joules_per_output_token is present', () => { - const e = entry({ joules_per_output_token: 8.4 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredJPerOutputToken).toBeDefined(); - expect(point.measuredJPerOutputToken!.y).toBe(8.4); - }); it('derives J/query, Wh/query, and percent TDP from validated source fields', () => { const e = entry({ avg_power_w: 560, joules_per_successful_query: 1800 }); @@ -1019,14 +1088,6 @@ describe('createChartDataPoint measured power fields', () => { expect(point.measuredAvgPower!.y).toBe(0); }); - it('emits measuredJPerTotalToken when joules_per_total_token is present', () => { - const e = entry({ joules_per_total_token: 0.93 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredJPerTotalToken).toBeDefined(); - expect(point.measuredJPerTotalToken!.y).toBe(0.93); - expect(point.measuredJPerTotalToken!.roof).toBe(false); - }); - it('emits J/output and J/total independently — different denominators', () => { // 8k1k workload: J/output ≈ 9 × J/total (input is ~8x output, so output/total ≈ 1/9). const e = entry({ joules_per_output_token: 2.04, joules_per_total_token: 0.23 }); @@ -1051,30 +1112,6 @@ describe('createChartDataPoint measured power fields', () => { // createChartDataPoint — per-stage measured power / energy (disagg prefill/decode) // =========================================================================== describe('createChartDataPoint per-stage measured power fields', () => { - it('emits measuredPrefillAvgPower when prefill_avg_power_w is present', () => { - const e = entry({ prefill_avg_power_w: 920.3 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredPrefillAvgPower).toBeDefined(); - expect(point.measuredPrefillAvgPower!.y).toBe(920.3); - expect(point.measuredPrefillAvgPower!.roof).toBe(false); - }); - - it('emits measuredDecodeAvgPower when decode_avg_power_w is present', () => { - const e = entry({ decode_avg_power_w: 612.1 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredDecodeAvgPower).toBeDefined(); - expect(point.measuredDecodeAvgPower!.y).toBe(612.1); - expect(point.measuredDecodeAvgPower!.roof).toBe(false); - }); - - it('emits measuredJPerInputToken when joules_per_input_token is present', () => { - const e = entry({ joules_per_input_token: 0.27 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredJPerInputToken).toBeDefined(); - expect(point.measuredJPerInputToken!.y).toBe(0.27); - expect(point.measuredJPerInputToken!.roof).toBe(false); - }); - it('omits all per-stage fields on legacy rows predating per-stage attribution', () => { // Single-node / pre-disagg runs emit avg_power_w only, no prefill/decode split. const e = entry({ avg_power_w: 685.5 }); @@ -1084,15 +1121,6 @@ describe('createChartDataPoint per-stage measured power fields', () => { expect(point.measuredJPerInputToken).toBeUndefined(); }); - it('emits prefill and decode independently — the disagg per-stage split', () => { - // GB300 disagg: prefill GPUs run compute-bound (higher W) than decode GPUs. - const e = entry({ prefill_avg_power_w: 948, decode_avg_power_w: 631 }); - const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); - expect(point.measuredPrefillAvgPower!.y).toBe(948); - expect(point.measuredDecodeAvgPower!.y).toBe(631); - expect(point.measuredPrefillAvgPower!.y).toBeGreaterThan(point.measuredDecodeAvgPower!.y); - }); - it('preserves a zero per-stage power value (not falsy-coerced away)', () => { // Same typeof===number gate as total power — 0 W must survive, not be dropped. const e = entry({ prefill_avg_power_w: 0, decode_avg_power_w: 0 }); diff --git a/packages/app/src/lib/chart-utils.ts b/packages/app/src/lib/chart-utils.ts index e94032342..78b62b2d8 100644 --- a/packages/app/src/lib/chart-utils.ts +++ b/packages/app/src/lib/chart-utils.ts @@ -11,6 +11,7 @@ import type { AggDataEntry, ChartDefinition, InferenceData, + PowerBasisFieldKey, YAxisMetricKey, } from '@/components/inference/types'; import { @@ -20,6 +21,8 @@ import { import { DEFAULT_TCO_BASIS, getGpuSpecs, isKnownGpu, type TcoBasis } from '@/lib/constants'; import { getVendor, type Vendor } from '@/lib/dynamic-colors'; import type { Locale } from '@/lib/i18n'; +import { buildPowerBasisChartFields, type PowerBasisChartFields } from '@/lib/power-basis'; +import { reconstructedRoleEnergy } from '@/components/inference/utils/role-energy'; // --------------------------------------------------------------------------- // High-contrast color generation (iwanthue — k-means in CIELab) @@ -289,7 +292,13 @@ export function buildAvailabilityHwKey( return hwKey; } -export type DerivedMetricKey = BenchmarkMetricKey; +// Power-boundary fields are derived here before the registry exposes them as +// axes; the union collapses once METRIC_REGISTRY carries the same keys. The +// reconstructed prefill energy is a comparison-only series (never an axis). +export type DerivedMetricKey = + | BenchmarkMetricKey + | PowerBasisFieldKey + | 'reconstructedPrefillJPerOutputToken'; export type DerivedChartFields = Pick; const chartMetric = (y: number): { y: number; roof: boolean } => ({ y, roof: false }); @@ -411,6 +420,11 @@ export function buildDerivedChartFields( hardwarePower && tputPerGpu ? (hardwarePower * 1000) / tputPerGpu : 0, ); } + // jOutput keeps the historical per-GPU normalization: for disaggregated rows + // output_tput_per_gpu is per decode GPU, so this is all-in W of one decode GPU + // per output token and ignores the prefill pool. The power-boundary field + // utilityProvisionedJPerOutputToken uses the same all-in W but counts every + // allocated GPU, so the two differ on disaggregated rows by (P + D) / D. if (hardwarePower > 0 && wants('jOutput') && outputTputPerGpu) { fields.jOutput = chartMetric(hardwarePower ? (hardwarePower * 1000) / outputTputPerGpu : 0); } @@ -426,6 +440,14 @@ export function buildDerivedChartFields( if (wants(key)) fields[key] = value; } + const powerBasis = buildPowerBasisChartFields(entry, specs); + for (const [key, value] of Object.entries(powerBasis) as [ + keyof PowerBasisChartFields, + { y: number; roof: boolean }, + ][]) { + if (wants(key)) fields[key] = value; + } + if (wants('modeledChassisPowerPerGpu') && entry.modeledSystemPower?.status === 'supported') { fields.modeledChassisPowerPerGpu = chartMetric(entry.modeledSystemPower.chassisAcWattsPerGpu); } @@ -521,6 +543,8 @@ type MeasuredPowerChartFields = Partial< | 'measuredJPerSuccessfulQuery' | 'measuredWhPerSuccessfulQuery' | 'measuredPowerPercentTdp' + | 'measuredPowerTimeline' + | 'reconstructedPrefillJPerOutputToken' > >; @@ -530,8 +554,13 @@ function buildMeasuredPowerChartFields( tdpWatts: number, ): MeasuredPowerChartFields { return { + // The timeline axis aliases the validated average: the point set (and + // its table row) is the same, only the chart body changes. ...(typeof entry.avg_power_w === 'number' - ? { measuredAvgPower: chartMetric(entry.avg_power_w) } + ? { + measuredAvgPower: chartMetric(entry.avg_power_w), + measuredPowerTimeline: chartMetric(entry.avg_power_w), + } : {}), ...(typeof entry.p75_power_w === 'number' && Number.isFinite(entry.p75_power_w) ? { measuredP75Power: chartMetric(entry.p75_power_w) } @@ -562,6 +591,14 @@ function buildMeasuredPowerChartFields( ...(typeof entry.decode_joules_per_output_token === 'number' ? { measuredDecodeJPerOutputToken: chartMetric(entry.decode_joules_per_output_token) } : {}), + // Prefill energy on the output-token axis, so the roles comparison can + // stack it against the decode pool (PowerX Figure 7). + ...(() => { + const roleEnergy = reconstructedRoleEnergy(entry); + return roleEnergy + ? { reconstructedPrefillJPerOutputToken: chartMetric(roleEnergy.prefill) } + : {}; + })(), ...(typeof entry.joules_per_successful_query === 'number' ? { measuredJPerSuccessfulQuery: chartMetric(entry.joules_per_successful_query), diff --git a/packages/app/src/lib/csv-export-helpers.test.ts b/packages/app/src/lib/csv-export-helpers.test.ts index 5fe616931..316f6af1f 100644 --- a/packages/app/src/lib/csv-export-helpers.test.ts +++ b/packages/app/src/lib/csv-export-helpers.test.ts @@ -8,6 +8,7 @@ import { historicalTrendToCsv, } from './csv-export-helpers'; import type { InferenceData } from '@/components/inference/types'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; const makePoint = (overrides: Partial = {}): InferenceData => ({ x: 100, @@ -752,3 +753,39 @@ describe('historicalTrendToCsv (mirrors HistoricalTrendsDisplay export)', () => expect(rows[1][headers.indexOf('Date')]).toBe('2025-01-15'); }); }); + +describe('inferenceChartToCsv power comparison', () => { + it.each([false, true])('exports plotted role values with overlay=%s', (overlay) => { + const base = makePoint({ + hwKey: 'gb300_dynamo-trt', + y: 708.1, + measuredAvgPower: { y: 708.1, roof: false }, + measuredPrefillAvgPower: { y: 760.442, roof: false }, + measuredDecodeAvgPower: { y: 690.652, roof: false }, + run_url: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35532106109', + }); + const points = expandPowerCompareSeries([base], 'y_measuredAvgPower', 'roles'); + const { headers, rows } = inferenceChartToCsv( + overlay ? [] : points, + 'Kimi-K3', + 'agentic-traces', + overlay ? points : [], + { + yHeader: 'Measured Power per Chip (W)', + yPath: 'measuredAvgPower.y', + xHeader: 'Interactivity (tok/s/user)', + }, + ); + expect( + rows.map((row) => [ + row[headers.indexOf('Power Series')], + row[headers.indexOf('Measured Power per Chip (W)')], + ]), + ).toEqual([ + ['All GPUs', 708.1], + ['Prefill GPUs', 760.442], + ['Decode GPUs', 690.652], + ]); + expect(points.every((point) => point.measuredAvgPower?.y === 708.1)).toBe(true); + }); +}); diff --git a/packages/app/src/lib/csv-export-helpers.ts b/packages/app/src/lib/csv-export-helpers.ts index 576569af8..a58ba948f 100644 --- a/packages/app/src/lib/csv-export-helpers.ts +++ b/packages/app/src/lib/csv-export-helpers.ts @@ -9,6 +9,7 @@ import { METRIC_REGISTRY } from '@/components/inference/metric-registry'; import type { InferenceData, TrendDataPoint } from '@/components/inference/types'; +import { inferPowerCompare, powerSeriesLabel } from '@/components/inference/utils/power-compare'; import { chipCounts } from '@/lib/chip-counts'; import type { SubmissionVolumeRow } from '@/lib/submissions-types'; @@ -57,6 +58,12 @@ export function inferenceChartToCsv( const islOsl = sequenceToIslOsl(sequence); const showModeledPower = displayedMetrics?.yPath === METRIC_REGISTRY.modeledChassisPowerPerGpu.field; + // A power comparison (`i_pcompare`) appends boundary / role clones of the + // plotted points; name each row's series so the export stays unambiguous. + const allPoints = [...data, ...overlayData]; + const powerCompare = inferPowerCompare(allPoints); + const showPowerSeries = powerCompare !== 'none'; + const plottedMetric = displayedMetrics ? `y_${displayedMetrics.yPath.split('.')[0]}` : ''; const headers = [ 'Model', 'ISL', @@ -111,13 +118,15 @@ export function inferenceChartToCsv( 'Physical Chips', 'DP', ...(showModeledPower ? ['Configured Chip Count'] : []), + ...(showPowerSeries ? ['Power Series'] : []), ]; const displayedColumns = displayedMetrics ? [ { header: displayedMetrics.yHeader, - value: (point: InferenceData) => nestedMetric(point, displayedMetrics.yPath), + value: (point: InferenceData) => + point.powerVariant ? point.y : nestedMetric(point, displayedMetrics.yPath), }, { header: displayedMetrics.xHeader, value: (point: InferenceData) => point.x }, ].filter( @@ -128,7 +137,7 @@ export function inferenceChartToCsv( : []; headers.splice(10, 0, ...displayedColumns.map((column) => column.header)); - const rows = [...data, ...overlayData] + const rows = allPoints .filter((d) => !d.hidden) .map((d) => { const chips = chipCounts(d, showModeledPower); @@ -177,6 +186,7 @@ export function inferenceChartToCsv( chips.physical, d.dp ?? '', ...(showModeledPower ? [chips.configured] : []), + ...(showPowerSeries ? [powerSeriesLabel(d, plottedMetric, powerCompare, 'en')] : []), ]; row.splice(10, 0, ...displayedColumns.map((column) => column.value(d))); return row; diff --git a/packages/app/src/lib/inference-labels.ts b/packages/app/src/lib/inference-labels.ts index c95883095..7854d7d41 100644 --- a/packages/app/src/lib/inference-labels.ts +++ b/packages/app/src/lib/inference-labels.ts @@ -41,6 +41,48 @@ export function inferenceFrameworkLabelOverride( } /** Keep unofficial-run identity/markers while making its special engine visible. */ +/** Leads every unofficial-run label, in line labels and legend rows alike. */ +export const OVERLAY_LABEL_MARKER = '✕ '; +const RUN_TAG_MAX = 20; +const RUN_TAG_TAIL = 17; + +export interface OverlayRunIdentity { + id: number | string; + branch?: string | null; +} + +/** + * Short, still recognisable name for a run: the whole branch when it is short, + * else its last path segment, else the branch tail (klaud nightlies end in + * `-`). Falls back to the run id when the branch is unknown. + */ +export function shortRunTag(run: OverlayRunIdentity): string { + const branch = run.branch?.trim() || `run ${run.id}`; + if (branch.length <= RUN_TAG_MAX) return branch; + const segment = branch.slice(branch.lastIndexOf('/') + 1); + if (segment.length > 0 && segment.length <= RUN_TAG_MAX) return segment; + return `…${branch.slice(-RUN_TAG_TAIL)}`; +} + +/** ` · ` appended to an overlay line label when other runs draw the same hardware. */ +export function overlayRunTag(run: OverlayRunIdentity): string { + return ` · ${shortRunTag(run)}`; +} + +/** + * Line-label text for an unofficial-run curve. Pills name the hardware, not the + * branch: branch names run to 70+ characters and the legend already carries + * them. The run tag is added only when several overlay runs draw the same + * hardware, so the pills stay distinguishable. + */ +export function getOverlayLineLabel( + hardwareLabel: string, + run: OverlayRunIdentity, + sharesHardware: boolean, +): string { + return `${OVERLAY_LABEL_MARKER}${hardwareLabel}${sharesHardware ? overlayRunTag(run) : ''}`; +} + export function getInferenceRunLabel( label: string, points: readonly (RunProvenance & { framework?: string })[], diff --git a/packages/app/src/lib/modeled-system-power.test.ts b/packages/app/src/lib/modeled-system-power.test.ts index c968537ae..c8fb4c26e 100644 --- a/packages/app/src/lib/modeled-system-power.test.ts +++ b/packages/app/src/lib/modeled-system-power.test.ts @@ -222,6 +222,89 @@ describe('modeled system power admission and accounting', () => { expect(modelSystemPower(source)).toMatchObject({ reason: 'role-power' }); }); + it('models an aggregate multinode deployment without worker telemetry at the deployment mean', () => { + // Kimi K3 B200 dynamo-vLLM TP8/PP2 (prod rows, 2026-09-17): two eight-GPU hosts, + // aggregate producer, no per-worker array. The K3 H200 vLLM row below is + // TP16 × 2 DP replicas across four hosts. + const b200 = row({ + is_multinode: true, + num_prefill_gpu: 16, + num_decode_gpu: 16, + decode_num_workers: 1, + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 715.095, + avg_total_gpu_power_w: 11441.513, + decode_pp: 2, + }, + }); + const perChassis = estimateChassisPower('b200', 11441.513 / 2, 1.3)!; + const estimate = modelSystemPower(b200); + expect(estimate).toMatchObject({ + status: 'supported', + topologyBasis: 'uniform-hosts', + chassisBasis: 'full', + gpuCount: 16, + chassisCount: 2, + modeledGpuCount: 16, + }); + if (estimate.status !== 'supported') throw new Error('unreachable'); + expect(estimate.chassisAcWatts).toBeCloseTo(perChassis.chassisAcWatts * 2, 6); + expect(estimate.deploymentFacilityWatts).toBe(estimate.facilityWatts); + expect(estimate.chassisAcWattsPerGpu).toBeCloseTo(perChassis.chassisAcWatts / 8, 6); + + const h200 = row({ + hardware: 'h200', + is_multinode: true, + prefill_tp: 16, + decode_tp: 16, + decode_ep: 32, + decode_dp_attention: true, + decode_num_workers: 2, + num_prefill_gpu: 32, + num_decode_gpu: 32, + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 167.357, + avg_total_gpu_power_w: 5355.413, + decode_pp: 1, + }, + }); + expect(modelSystemPower(h200)).toMatchObject({ + status: 'supported', + topologyBasis: 'uniform-hosts', + chassisCount: 4, + gpuCount: 32, + }); + + // A replica count that does not explain the telemetry width is not guessed around. + expect(modelSystemPower({ ...h200, decode_num_workers: 1 })).toMatchObject({ + reason: 'gpu-count', + }); + // Twelve GPUs cannot fill whole eight-GPU hosts; placement is unknown. + const twelve = row({ is_multinode: true, decode_tp: 12, prefill_tp: 12 }); + twelve.metrics.avg_total_gpu_power_w = twelve.metrics.avg_power_w * 12; + expect(modelSystemPower(twelve)).toMatchObject({ reason: 'topology' }); + // Per-worker telemetry, when present, keeps the more exact worker path. + const withWorkers = { + ...b200, + workers: ['host-a', 'host-b'].map((host, worker_idx) => ({ + role: 'agg', + worker_idx, + hosts: [host], + num_gpus: 8, + avg_power_w: 715.095, + })), + }; + expect(modelSystemPower(withWorkers)).toMatchObject({ + status: 'supported', + topologyBasis: 'worker-hosts', + chassisCount: 2, + }); + }); + it('preserves meaningful aggregate PP and PCP aliases before checking physical width', () => { for (const widths of [ { decode_pp: 1, prefill_pp: 2 }, diff --git a/packages/app/src/lib/modeled-system-power.ts b/packages/app/src/lib/modeled-system-power.ts index 550fe23fc..aadb737f6 100644 --- a/packages/app/src/lib/modeled-system-power.ts +++ b/packages/app/src/lib/modeled-system-power.ts @@ -43,7 +43,13 @@ export type SystemPowerEstimate = deploymentFacilityWatts: number; pue: number; telemetryBasis: 'validated-v2' | 'validated-unversioned-single-node'; - topologyBasis: 'single-node' | 'worker-hosts'; + /** + * 'single-node': one host, one chassis. 'worker-hosts': one chassis per + * measured worker, each at its own telemetry. 'uniform-hosts': an + * aggregate multinode deployment whose producer emitted no per-worker + * telemetry; every eight-GPU chassis is modeled at the deployment mean. + */ + topologyBasis: 'single-node' | 'worker-hosts' | 'uniform-hosts'; /** * 'full': every chassis had all eight GPUs measured. 'extrapolated': at least * one chassis was partially allocated; its model input is the measured per-GPU @@ -138,7 +144,7 @@ export function modelSystemPower( } const chassis: MeasuredChassis[] = []; - let topologyBasis: 'single-node' | 'worker-hosts'; + let topologyBasis: 'single-node' | 'worker-hosts' | 'uniform-hosts'; if (row.disagg === false && row.is_multinode === false) { // One host cannot hold more than one chassis. if (gpuCount > CHASSIS_GPU_COUNT) return unavailable('topology'); @@ -173,6 +179,36 @@ export function modelSystemPower( ? m.avg_total_gpu_power_w : m.avg_power_w * CHASSIS_GPU_COUNT, }); + } else if (row.disagg === false && (!Array.isArray(row.workers) || row.workers.length === 0)) { + // Aggregate multinode producers emit no per-worker telemetry. Symmetric + // TP/PP/DP shards load every host alike, so each full eight-GPU chassis is + // modeled at the deployment mean; the supported hardware only ships in + // eight-GPU hosts, so the count must fill whole chassis on several hosts. + // Disaggregated roles differ in load and stay on the worker path. + const hostCount = gpuCount / CHASSIS_GPU_COUNT; + if (!count(hostCount) || hostCount < 2) return unavailable('topology'); + const tp = row.decode_tp > 0 ? row.decode_tp : row.prefill_tp; + const pp = Math.max(m.pp ?? 1, m.decode_pp ?? 1, m.prefill_pp ?? 1); + const pcp = Math.max(m.pcp_size ?? 1, m.decode_pcp_size ?? 1, m.prefill_pcp_size ?? 1); + // Data-parallel replicas widen the deployment beyond one TP×PP×PCP group. + const replicas = Math.max(1, row.decode_num_workers); + if ( + !count(tp) || + !count(pp) || + !count(pcp) || + !count(replicas) || + tp * pp * pcp * replicas !== gpuCount + ) { + return unavailable('gpu-count'); + } + topologyBasis = 'uniform-hosts'; + for (let host = 0; host < hostCount; host++) { + chassis.push({ + measuredGpus: CHASSIS_GPU_COUNT, + // Partition the producer's exact total so the chassis inputs sum back to it. + modelInputWatts: m.avg_total_gpu_power_w / hostCount, + }); + } } else { // A role average across several hosts is insufficient for nonlinear // fan/PSU evaluation. Require one chassis per measured worker and a diff --git a/packages/app/src/lib/power-basis.test.ts b/packages/app/src/lib/power-basis.test.ts new file mode 100644 index 000000000..0aa67a4f5 --- /dev/null +++ b/packages/app/src/lib/power-basis.test.ts @@ -0,0 +1,73 @@ +import { describe, expect, it } from 'vitest'; + +import type { BenchmarkRow } from '@/lib/api'; +import { rowToAggDataEntry, transformBenchmarkRows } from '@/lib/benchmark-transform'; +import { buildDerivedChartFields, getHardwareKey } from '@/lib/chart-utils'; +import { POWER_BASIS_FIELDS } from '@/lib/power-basis'; + +// Qwen3.5 B200 c1, run 34175132645: actual rounded telemetry, eight GPUs +// (same fixture as modeled-system-power.test.ts) plus an output rate. +function row(overrides: Partial = {}): BenchmarkRow { + return { + id: 441192, + model: 'qwen3.5', + hardware: 'b200', + framework: 'sglang', + precision: 'fp8', + spec_method: 'none', + disagg: false, + is_multinode: false, + prefill_tp: 8, + prefill_ep: 1, + prefill_dp_attention: false, + prefill_num_workers: 0, + decode_tp: 8, + decode_ep: 1, + decode_dp_attention: false, + decode_num_workers: 0, + num_prefill_gpu: 8, + num_decode_gpu: 8, + benchmark_type: 'single_turn', + isl: 8192, + osl: 1024, + conc: 1, + offload_mode: 'off', + image: 'lmsysorg/sglang:v0.5.19-cu130', + date: '2026-09-08', + run_url: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34175132645/attempts/1', + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 349.859, + avg_total_gpu_power_w: 2798.868, + total_gpu_energy_j: 120361.299, + joules_per_output_token: 12.937902, + tput_per_gpu: 270, + output_tput_per_gpu: 30, + pp: 1, + pcp_size: 1, + median_intvty: 100, + }, + ...overrides, + }; +} + +/** Full official-path derivation for one row with real HW_REGISTRY specs. */ +function derive(source: BenchmarkRow) { + const entry = rowToAggDataEntry(source); + const hwKey = getHardwareKey(entry); + return { entry, hwKey, fields: buildDerivedChartFields(entry, hwKey) }; +} + +describe('power boundaries through the derived-field builder', () => { + it('serves the same fields to ?unofficialrun= overlays through transformBenchmarkRows', () => { + const { chartData } = transformBenchmarkRows([row()], 'median', 'external'); + const point = chartData[0][0]; + const official = derive(row()).fields; + for (const basis of Object.values(POWER_BASIS_FIELDS)) { + expect(point[basis.watts]).toEqual(official[basis.watts]); + expect(point[basis.energy]).toEqual(official[basis.energy]); + expect(point[basis.watts]?.roof).toBe(false); + } + }); +}); diff --git a/packages/app/src/lib/power-basis.ts b/packages/app/src/lib/power-basis.ts new file mode 100644 index 000000000..034eeca2d --- /dev/null +++ b/packages/app/src/lib/power-basis.ts @@ -0,0 +1,211 @@ +/** + * Power boundaries for one benchmark point, from the GPU board out to the + * utility meter. Pure numbers only — no React, DOM, or registry imports — so + * the derived-field builder, overlays, and tests share one formula set. + * + * | Basis | W / GPU | J / output token | + * | ---------------------- | ------------------------------------------------ | ----------------------------------------- | + * | B1 gpu-measured | `avg_power_w` (existing `measuredAvgPower`) | `joules_per_output_token` (existing) | + * | B2 gpu-provisioned | `HW_REGISTRY.tdp` | W × N_alloc ÷ total output tok/s | + * | B3 utility-provisioned | `HW_REGISTRY.power` × 1000 | W × N_alloc ÷ total output tok/s | + * | B4 utility-modeled | modeled `deploymentFacilityWatts` ÷ `gpuCount` | B1 J/out × (B4 W ÷ B1 W) | + * + * N_alloc counts every allocated GPU (prefill + decode for disaggregation); + * total output tok/s is the whole deployment's. B4 reuses the estimate that + * `modelSystemPower` already attached to the entry — PUE is applied exactly + * once inside that model, never here — and is withheld wherever B1 is. + * Unavailable values are `null`; callers omit the field rather than plotting 0. + */ +import type { AggDataEntry, InferenceData, PowerBasisFieldKey } from '@/components/inference/types'; + +export const POWER_BASES = [ + 'gpu-measured', + 'gpu-provisioned', + 'utility-provisioned', + 'utility-modeled', +] as const; +export type PowerBasis = (typeof POWER_BASES)[number]; +export type PowerQuantity = 'watts' | 'energy'; + +export const POWER_BASIS_LABELS: Record = { + 'gpu-measured': { en: 'GPU measured', zh: 'GPU 实测' }, + 'gpu-provisioned': { en: 'GPU provisioned (TDP)', zh: 'GPU 额定(TDP)' }, + 'utility-provisioned': { en: 'Utility provisioned (all-in)', zh: '全电源配置(all-in)' }, + 'utility-modeled': { en: 'Utility modeled (PUE)', zh: '数据中心建模(含 PUE)' }, +}; + +/** InferenceData keys per derived basis and quantity. B1 lives on the measured* fields. */ +export const POWER_BASIS_FIELDS: Record< + Exclude, + Record +> = { + 'gpu-provisioned': { + watts: 'gpuProvisionedWatts', + energy: 'gpuProvisionedJPerOutputToken', + }, + 'utility-provisioned': { + watts: 'utilityProvisionedWatts', + energy: 'utilityProvisionedJPerOutputToken', + }, + 'utility-modeled': { + watts: 'utilityModeledWatts', + energy: 'utilityModeledJPerOutputToken', + }, +}; + +export interface PowerBasisInput { + /** B2 W/GPU: HW_REGISTRY tdp. 0 means the spec is not yet available. */ + tdpWatts: number | null; + /** B3 W/GPU: HW_REGISTRY all-in power, already in watts. */ + utilityWatts: number | null; + /** Every GPU the deployment occupies (prefill + decode for disaggregation). */ + allocatedGpus: number | null; + /** Whole-deployment successful output tokens per second. */ + totalOutputTokPerSec: number | null; + /** B1 W/GPU from validated telemetry. */ + measuredWatts: number | null; + /** B1 J/output token from the same telemetry window. */ + measuredJPerOutputToken: number | null; + /** + * B4 W/GPU: modeled facility watts (PUE already applied) per measured GPU. + * B4 is a scaling of B1, so it is withheld whenever `measuredWatts` is null. + */ + modeledFacilityWattsPerGpu: number | null; +} + +export type PowerBasisValues = Record; + +const positive = (value: unknown): value is number => + typeof value === 'number' && Number.isFinite(value) && value > 0; +const count = (value: unknown): value is number => positive(value) && Number.isSafeInteger(value); +const orNull = (value: number): number | null => (positive(value) ? value : null); + +/** + * Derives the B2–B4 boundary values from plain numbers. Any unavailable input + * yields `null` for the values that depend on it and leaves the rest intact. + */ +export function computePowerBasisFields(input: PowerBasisInput): PowerBasisValues { + const tdp = positive(input.tdpWatts) ? input.tdpWatts : null; + const utility = positive(input.utilityWatts) ? input.utilityWatts : null; + const measuredWatts = positive(input.measuredWatts) ? input.measuredWatts : null; + const measuredJ = positive(input.measuredJPerOutputToken) ? input.measuredJPerOutputToken : null; + // B4 is B1 carried out to the utility meter, so it follows B1's availability: + // no measured watts, no modeled boundary (B3 ≥ B4 ≥ B1 needs its anchor). + const modeled = + measuredWatts !== null && positive(input.modeledFacilityWattsPerGpu) + ? input.modeledFacilityWattsPerGpu + : null; + + // Provisioned energy: GPU-seconds spent per output token by the whole + // deployment (N_alloc ÷ total tok/s) × W per GPU = J per output token. + const gpuSecondsPerOutputToken = + positive(input.allocatedGpus) && positive(input.totalOutputTokPerSec) + ? input.allocatedGpus / input.totalOutputTokPerSec + : null; + const provisionedEnergy = (watts: number | null) => + watts !== null && gpuSecondsPerOutputToken !== null + ? orNull(watts * gpuSecondsPerOutputToken) + : null; + + // Modeled energy scales the producer's same-window E/N by modeled ÷ measured W, + // so it inherits B1's token denominator instead of re-deriving one. + const modeledEnergy = + modeled !== null && measuredWatts !== null && measuredJ !== null + ? orNull((measuredJ * modeled) / measuredWatts) + : null; + + return { + gpuProvisionedWatts: tdp, + gpuProvisionedJPerOutputToken: provisionedEnergy(tdp), + utilityProvisionedWatts: utility, + utilityProvisionedJPerOutputToken: provisionedEnergy(utility), + utilityModeledWatts: modeled, + utilityModeledJPerOutputToken: modeledEnergy, + }; +} + +type PowerBasisEntry = Pick< + AggDataEntry, + | 'output_tput_per_gpu' + | 'disagg' + | 'benchmark_type' + | 'num_prefill_gpu' + | 'num_decode_gpu' + | 'avg_power_w' + | 'joules_per_output_token' + | 'modeledSystemPower' +>; + +/** + * Whole-deployment normalization for the provisioned energies. Aggregate rows + * already report output per allocated GPU, so N_alloc cancels and the ratio + * 1 GPU : per-GPU throughput is exact without trusting display counts (legacy + * ingest can encode TP × EP twice). Fixed-sequence disaggregated rows report + * output per decode GPU while the deployment also powers the prefill pool, so + * total output = per-GPU × decode GPUs and N_alloc = prefill + decode GPUs. + * Other disaggregated benchmark types are left out: whether AgentX throughput + * already divides by all GPUs is not verifiable in-app. + */ +export function powerBasisNormalization( + entry: Pick< + PowerBasisEntry, + 'output_tput_per_gpu' | 'disagg' | 'benchmark_type' | 'num_prefill_gpu' | 'num_decode_gpu' + >, +): Pick { + const perGpu = entry.output_tput_per_gpu; + const unavailable = { allocatedGpus: null, totalOutputTokPerSec: null }; + if (!positive(perGpu)) return unavailable; + if (!entry.disagg) return { allocatedGpus: 1, totalOutputTokPerSec: perGpu }; + if (entry.benchmark_type !== 'single_turn') return unavailable; + const prefill = entry.num_prefill_gpu; + const decode = entry.num_decode_gpu; + if (!count(prefill) || !count(decode)) return unavailable; + return { allocatedGpus: prefill + decode, totalOutputTokPerSec: perGpu * decode }; +} + +/** + * B4 W/GPU from the estimate `rowToAggDataEntry` attached. The model owns + * telemetry admission: `modelSystemPower` requires `power_valid === 1` plus + * schema v2, or the validated unversioned single-node producer it records as + * `telemetryBasis: 'validated-unversioned-single-node'`. That is the same + * population the app plots as B1 (`measuredAvgPower`) and as + * `modeledChassisPowerPerGpu`, so B4 renders exactly where they do. The public + * API's stricter `strictV2` row filter is not re-applied here; it is not + * applied to the chart's B1 either. + */ +export function modeledFacilityWattsPerGpu( + entry: Pick, +): number | null { + const model = entry.modeledSystemPower; + if (model?.status !== 'supported') return null; + if (!positive(model.deploymentFacilityWatts) || !count(model.gpuCount)) return null; + return orNull(model.deploymentFacilityWatts / model.gpuCount); +} + +export type PowerBasisChartFields = Partial>; + +/** + * Chart-shaped B2–B4 fields for one entry. Keys are present only for finite, + * positive values: the metric filters drop a point by `metricKey in point`, + * and the coordinate remap falls back to raw throughput when a key exists + * with an unusable value. + */ +export function buildPowerBasisChartFields( + entry: PowerBasisEntry, + specs: { tdp?: number; power?: number }, +): PowerBasisChartFields { + const values = computePowerBasisFields({ + tdpWatts: specs.tdp ?? null, + utilityWatts: positive(specs.power) ? specs.power * 1000 : null, + ...powerBasisNormalization(entry), + measuredWatts: entry.avg_power_w ?? null, + measuredJPerOutputToken: entry.joules_per_output_token ?? null, + modeledFacilityWattsPerGpu: modeledFacilityWattsPerGpu(entry), + }); + const fields: PowerBasisChartFields = {}; + for (const key of Object.keys(values) as PowerBasisFieldKey[]) { + const y = values[key]; + if (y !== null) fields[key] = { y, roof: false }; + } + return fields; +} diff --git a/packages/app/src/lib/url-state.ts b/packages/app/src/lib/url-state.ts index da7bb00d3..78755aa95 100644 --- a/packages/app/src/lib/url-state.ts +++ b/packages/app/src/lib/url-state.ts @@ -63,6 +63,12 @@ const URL_STATE_KEYS = [ 'i_spec', // Measured-power certification tiers ('certified' / 'legacy', comma-joined). 'i_power', + // Completed Perf Rulers on the primary inference chart: `isoX|curveA|curveB` + // entries joined by `;` (see serializePerfRulers in d3-chart/layers/perf-ruler). + 'i_rulers', + // Comparison series overlaid on a gated power metric: `boundaries` (every + // power boundary) or `roles` (prefill / decode pools). Empty = the metric alone. + 'i_pcompare', // Exact serving-envelope pair behind an Overview 30-day comparison cell. 'i_overview_current', 'i_overview_baseline', @@ -177,6 +183,8 @@ export const PARAM_DEFAULTS: Record = { i_disagg: '', i_spec: '', i_power: '', + i_rulers: '', + i_pcompare: '', i_overview_current: '', i_overview_baseline: '', e_rundate: '', @@ -492,6 +500,26 @@ export function rememberChartStateInUrl(): string { return chartParams.toString(); } +/** + * The current page's URL carrying its chart state plus `overrides`, + * canonicalised like `rememberChartStateInUrl`: chart params and both + * unofficial-run spellings are dropped from the live address bar before the + * store's state (and the overrides) are layered on. For anchors that must + * work with open-in-new-tab, where the in-memory state would otherwise be lost. + */ +export function chartStateHref(overrides: Record): string { + const { origin, pathname, hash, search } = window.location; + const merged = new URLSearchParams(search); + for (const key of URL_STATE_KEYS) merged.delete(key); + // Collected first: deleting while iterating the params would skip entries. + const staleRunKeys = [...merged.keys()].filter((key) => UNOFFICIAL_RUN_PARAM_RE.test(key)); + for (const key of staleRunKeys) merged.delete(key); + for (const [key, value] of collectTabParams()) merged.set(key, value); + for (const [key, value] of Object.entries(overrides)) merged.set(key, value); + const query = merged.toString(); + return `${origin}${pathname}${query ? `?${query}` : ''}${hash}`; +} + /** * Append the current chart state to an outbound in-app href, so the page it * opens can link back to the chart the user left. Used for the agentic From 9096c00b2ac54d4333763125590c017818a7fd89 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Tue, 29 Sep 2026 12:27:46 -0700 Subject: [PATCH 2/7] feat(ui): persist perf rulers in share links MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Serialize curve-to-curve perf rulers into the i_rulers URL parameter and restore them on load; the InferenceContext and url-state wiring lives in the boundary-metrics commit. 中文:把曲线间 perf ruler 序列化到 i_rulers URL 参数并在加载时恢复;InferenceContext 与 url-state 的接线在边界指标提交中。 --- .../components/inference/perf-ruler-store.ts | 179 ++++++++++++++++++ .../lib/d3-chart/layers/perf-ruler.test.ts | 30 +++ .../app/src/lib/d3-chart/layers/perf-ruler.ts | 66 +++++++ 3 files changed, 275 insertions(+) create mode 100644 packages/app/src/components/inference/perf-ruler-store.ts diff --git a/packages/app/src/components/inference/perf-ruler-store.ts b/packages/app/src/components/inference/perf-ruler-store.ts new file mode 100644 index 000000000..98bbfd692 --- /dev/null +++ b/packages/app/src/components/inference/perf-ruler-store.ts @@ -0,0 +1,179 @@ +'use client'; + +import { + type Dispatch, + type SetStateAction, + createContext, + useCallback, + useContext, + useMemo, + useRef, + useState, +} from 'react'; + +import { track } from '@/lib/analytics'; +import { perfRulerAxisMetricKey } from '@/hooks/usePerfRulerAxisReset'; +import { + EMPTY_PERF_RULER_STATE, + MAX_PERF_RULERS, + type PerfRulerMeasurement, + type PerfRulerState, + clearPerfRulers, + parsePerfRulers, +} from '@/lib/d3-chart/layers/perf-ruler'; + +/** + * @file perf-ruler-store.ts + * @description Provider-owned Perf Ruler state for the primary inference + * chart, so completed rulers persist in share links (`i_rulers`). Lives + * beside `InferenceContext` rather than inside it so `ScatterGraph` can + * consume the store without importing the (heavily mocked) provider module. + */ + +/** + * The chart instance whose Perf Rulers persist in share links. `ChartDisplay` + * mounts the primary chart as `chart-${graphIndex}` and only graph 0 is ever + * visible; the replay chart (`replay-chart-0`) draws interpolated frames of + * the same curves and must NOT bind, or every ruler would render twice and + * the replay's prune pass could delete rulers the main chart still shows. + */ +export const PERSISTED_PERF_RULER_CHART_ID = 'chart-0'; + +/** + * Perf-ruler store for the persisted chart. Lives in its own context rather + * than the Display domain so a ruler commit does not rerender every display + * consumer, and so harnesses that mount `InferenceContextsProvider` with + * static mock values (no store) keep the chart's component-local fallback. + * + * `state` holds COMMITTED rulers: the D3 layer renders them and `i_rulers` + * serializes them. `pending` holds rulers parsed from the share link whose + * curves may not have been drawn yet — data, `i_gpus`, comparison dates, + * and `?unofficialrun=` overlays all arrive after the chart's first draw, + * and the chart prunes any committed ruler whose curve path is absent from + * the DOM. The chart therefore commits a pending ruler only once BOTH of + * its curve paths exist (see the perf-ruler decoration effect in + * ScatterGraph); rulers whose curves never appear stay pending, invisible + * and unserialized, until an axis change or an explicit clear discards them. + */ +export interface PerfRulerStore { + chartId: string; + state: PerfRulerState; + setState: Dispatch>; + pending: readonly PerfRulerMeasurement[] | null; + /** + * Commit share-link rulers whose curves now exist (`resolved`, iso-x + * already clamped to the pair's overlap) and keep `remaining` pending. + */ + commitPending: ( + resolved: readonly PerfRulerMeasurement[], + remaining: readonly PerfRulerMeasurement[] | null, + ) => void; + /** Drop share-link rulers that were never committed (toggle-off, clear). */ + discardPending: () => void; +} + +/** + * Axis identity of the chart `ChartDisplay` renders as `chart-0`, for the + * store's axis reset. `graphs` is always `[interactivity, e2e]`, but + * ChartDisplay shows the e2e graph for every non-interactivity x mode + * (`visibleGraphs`), so the rendered chart — not `graphs[0]` — is what the + * rulers were placed on. The x mode itself is part of the identity as well: + * the derived agentic modes (e2e-normalized interactivity, …) override the + * e2e graph's `x_scale_field` inside ChartDisplay only, so here the same + * definition still reads `_e2el` for those modes. Percentile changes + * are already encoded in `x_scale_field`. Null while no graph exists. + */ +export function persistedPerfRulerAxisKey( + graphs: readonly { chartDefinition: { chartType: string; x_scale_field: string } }[], + xAxisMode: string, + yAxisMetric: string, +): string | null { + const wantedType = xAxisMode === 'interactivity' ? 'interactivity' : 'e2e'; + const graph = + graphs.find((candidate) => candidate.chartDefinition.chartType === wantedType) ?? graphs[0]; + if (!graph) return null; + return perfRulerAxisMetricKey(`${xAxisMode}:${graph.chartDefinition.x_scale_field}`, yAxisMetric); +} + +/** Provided by `InferenceProvider`; exported for chart component tests. */ +export const PerfRulerStoreContext = createContext(undefined); + +/** The persisted-ruler store, or undefined outside `InferenceProvider`. */ +export function usePerfRulerStore(): PerfRulerStore | undefined { + return useContext(PerfRulerStoreContext); +} + +/** + * Owns the persisted perf-ruler state. Exported so component tests can host a + * real store around a chart without the full provider. + * + * `axisMetricKey` is the persisted chart's axis identity + * ({@link persistedPerfRulerAxisKey}), or null while no chart definition exists. + * The axis reset runs HERE, not through `usePerfRulerAxisReset` in the chart: + * that hook adjusts state during the chart's render, which is only legal for + * the chart's own state — updating a provider's state from a child's render + * is a React error. Same semantics: a change of either axis metric clears + * committed rulers (redrawn curves would give a ratio nobody placed) and + * discards pending ones (they were placed on the old axes). The null → key + * transition on first data is not a change, so share-link rulers survive + * the load; the x-mode fallback for fixed sequences also settles before any + * chart definition exists. + */ +export function usePerfRulerStoreValue( + chartId: string, + initialSerialized: string | undefined, + axisMetricKey: string | null, +): PerfRulerStore { + const [state, setState] = useState(EMPTY_PERF_RULER_STATE); + const [pending, setPending] = useState(() => { + const parsed = parsePerfRulers(initialSerialized).rulers; + return parsed.length > 0 ? parsed : null; + }); + // `interactivity_perf_ruler_shared_load` fires once per store — once per + // opened link — with the number of rulers the link carried, the first time + // any of them renders. Rulers commit per curve arrival (below), so a + // per-commit event would count one link several times with partial counts. + const linkRulerCountRef = useRef(pending?.length ?? 0); + const sharedLoadReportedRef = useRef(false); + + const [appliedAxisMetricKey, setAppliedAxisMetricKey] = useState(axisMetricKey); + if (axisMetricKey !== null && axisMetricKey !== appliedAxisMetricKey) { + setAppliedAxisMetricKey(axisMetricKey); + if (appliedAxisMetricKey !== null) { + setState(clearPerfRulers); + setPending(null); + } + } + + const commitPending = useCallback( + ( + resolved: readonly PerfRulerMeasurement[], + remaining: readonly PerfRulerMeasurement[] | null, + ) => { + if (resolved.length > 0) { + setState((prev) => { + // Fresh ids from the live counter: a ruler placed by hand before the + // share-link rulers resolved must keep its own join key. + const rulers = [ + ...prev.rulers, + ...resolved.map((ruler, index) => ({ ...ruler, id: prev.nextId + index })), + ]; + while (rulers.length > MAX_PERF_RULERS) rulers.shift(); + return { rulers, draft: prev.draft, nextId: prev.nextId + resolved.length }; + }); + if (!sharedLoadReportedRef.current) { + sharedLoadReportedRef.current = true; + track('interactivity_perf_ruler_shared_load', { count: linkRulerCountRef.current }); + } + } + setPending(remaining); + }, + [], + ); + const discardPending = useCallback(() => setPending(null), []); + + return useMemo( + () => ({ chartId, state, setState, pending, commitPending, discardPending }), + [chartId, state, pending, commitPending, discardPending], + ); +} diff --git a/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts b/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts index d50fe1573..c7cc0ad80 100644 --- a/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts +++ b/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts @@ -16,10 +16,12 @@ import { isPerfRulerCurveVisible, movePerfRulerIsoX, nextPerfRulerState, + parsePerfRulers, pathXExtent, perfRulerCurveSet, prunePerfRulers, renderPerfRulers, + serializePerfRulers, type PerfRulerEndInput, type PerfRulerGeometry, type PerfRulerLabelLayoutOptions, @@ -889,6 +891,34 @@ describe('perfRulerCurveSet', () => { }); }); +// ── serializePerfRulers / parsePerfRulers (share links) ───────────── + +describe('serializePerfRulers / parsePerfRulers', () => { + const OFFICIAL_A = 'roofline-b200_trt_fp8'; + const OFFICIAL_B = 'roofline-mi355x_sglang_fp4'; + // Overlay curves carry the unofficial run index; power-envelope curves are + // split per date with the encoded date appended (`%2F` from a slash). + const OVERLAY = 'overlay-roofline-h100_vllm_fp8_run1__2026-09%2F11'; + + it('round-trips completed rulers, including overlay and date-scoped curve ids', () => { + const state = complete( + complete(EMPTY_PERF_RULER_STATE, OFFICIAL_A, OFFICIAL_B, 41.5), + OFFICIAL_A, + OVERLAY, + 120, + ); + const encoded = serializePerfRulers(state); + expect(encoded).toBe(`41.5|${OFFICIAL_A}|${OFFICIAL_B};120|${OFFICIAL_A}|${OVERLAY}`); + const parsed = parsePerfRulers(encoded); + expect(parsed.rulers).toEqual([ + { id: 1, curveA: OFFICIAL_A, curveB: OFFICIAL_B, isoX: 41.5 }, + { id: 2, curveA: OFFICIAL_A, curveB: OVERLAY, isoX: 120 }, + ]); + expect(parsed.draft).toBeNull(); + expect(parsed.nextId).toBe(3); + }); +}); + // ── pathXExtent ───────────────────────────────────────────── describe('pathXExtent', () => { diff --git a/packages/app/src/lib/d3-chart/layers/perf-ruler.ts b/packages/app/src/lib/d3-chart/layers/perf-ruler.ts index c44143529..ef13b2a1d 100644 --- a/packages/app/src/lib/d3-chart/layers/perf-ruler.ts +++ b/packages/app/src/lib/d3-chart/layers/perf-ruler.ts @@ -323,6 +323,72 @@ export function prunePerfRulers( return { ...prev, rulers, draft }; } +const PERF_RULER_URL_RULER_SEPARATOR = ';'; +const PERF_RULER_URL_FIELD_SEPARATOR = '|'; +/** + * Shape of a curve id the link may reference: one roofline path's identity + * class (`roofline-` / `overlay-roofline-`), never the shared + * `roofline-path` / `overlay-roofline-path` marker classes or any other node + * inside the zoom group — those match many paths, so a hand-edited link + * would draw a ruler between whichever two come first in DOM order. + */ +const PERF_RULER_CURVE_ID = /^(?:overlay-)?roofline-(?!path$)[\w%.-]+$/u; + +/** + * Share-link encoding of the COMPLETED rulers (`i_rulers`). One ruler per + * `;`, fields joined by `|`: `isoX|curveA|curveB`. Curve ids are the rendered + * roofline path identity classes (`roofline-_`, + * `overlay-roofline-__run`, optionally `__`), whose alphabet is `[A-Za-z0-9_%.-]`, so neither separator can + * appear inside one; `URLSearchParams` percent-encodes both on the wire. + * The iso-x is rounded to four significant digits to keep links short — a + * 0.05% shift on the x metric is far below the ruler's visual resolution. + * The draft is never serialized: it is an unfinished click, not a + * measurement. Empty state serializes to '' so the param strips as default. + */ +export function serializePerfRulers(state: PerfRulerState): string { + return state.rulers + .map((ruler) => + [Number(ruler.isoX.toPrecision(4)), ruler.curveA, ruler.curveB].join( + PERF_RULER_URL_FIELD_SEPARATOR, + ), + ) + .join(PERF_RULER_URL_RULER_SEPARATOR); +} + +/** + * Inverse of {@link serializePerfRulers}. Malformed entries (wrong field + * count, non-numeric iso-x, identical curve ids, or ids that are not + * roofline identity classes) are dropped silently — a hand-edited or + * truncated link degrades to fewer rulers, never to an error. The list is capped at + * {@link MAX_PERF_RULERS} keeping the NEWEST (last-serialized) entries, the + * same end the click reducer drops from. Ids are reassigned 1..n with + * `nextId = n + 1`, so parsed rulers are valid D3 join keys and a ruler + * placed afterwards never collides. Returns {@link EMPTY_PERF_RULER_STATE} + * (same reference) for '', null, or an all-malformed value. + */ +export function parsePerfRulers(raw: string | null | undefined): PerfRulerState { + if (!raw) return EMPTY_PERF_RULER_STATE; + const parsed: Omit[] = []; + for (const entry of raw.split(PERF_RULER_URL_RULER_SEPARATOR)) { + const fields = entry.split(PERF_RULER_URL_FIELD_SEPARATOR); + if (fields.length !== 3) continue; + const [isoXField, curveA, curveB] = fields; + if (isoXField.trim() === '' || curveA === curveB) continue; + if (!PERF_RULER_CURVE_ID.test(curveA) || !PERF_RULER_CURVE_ID.test(curveB)) continue; + const isoX = Number(isoXField); + if (!Number.isFinite(isoX)) continue; + parsed.push({ curveA, curveB, isoX }); + } + if (parsed.length === 0) return EMPTY_PERF_RULER_STATE; + const kept = parsed.slice(-MAX_PERF_RULERS); + return { + rulers: kept.map((ruler, index) => ({ id: index + 1, ...ruler })), + draft: null, + nextId: kept.length + 1, + }; +} + /** Every curve referenced by any ruler or the draft (hit-halo styling). */ export function perfRulerCurveSet(state: PerfRulerState): Set { const curves = new Set(); From 10d99016d438d5000474baa57fdfac4e5953af79 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Tue, 29 Sep 2026 12:27:46 -0700 Subject: [PATCH 3/7] feat(ui): per-point PowerX view on the agentic detail page MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Open a benchmark point's stored telemetry from the chart tooltip: a PowerX tab on the agentic detail page and a dialog from the scatter/GPU charts, with per-chip toggles, a server-metric overlay and loading/error/missing states that keep cached data on a failed refetch. 中文:从图表 tooltip 打开 benchmark 点的入库遥测:agentic 详情页的 PowerX 标签页与散点/GPU 图的对话框,含逐芯片开关、server 指标叠加,以及刷新失败时保留缓存数据的加载/错误/缺失状态。 --- .../cypress/component/power-telemetry.cy.tsx | 205 ++++++++ .../agentic-point/agentic-point-detail.tsx | 6 + .../agentic-point/overlay-sources.ts | 83 ++++ .../agentic-point/power-telemetry-view.tsx | 448 ++++++++++++++++++ .../agentic-point/use-detail-view.ts | 8 +- .../inference/power-telemetry-dialog.tsx | 51 ++ 6 files changed, 799 insertions(+), 2 deletions(-) create mode 100644 packages/app/cypress/component/power-telemetry.cy.tsx create mode 100644 packages/app/src/components/inference/agentic-point/overlay-sources.ts create mode 100644 packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx create mode 100644 packages/app/src/components/inference/power-telemetry-dialog.tsx diff --git a/packages/app/cypress/component/power-telemetry.cy.tsx b/packages/app/cypress/component/power-telemetry.cy.tsx new file mode 100644 index 000000000..332a5fde4 --- /dev/null +++ b/packages/app/cypress/component/power-telemetry.cy.tsx @@ -0,0 +1,205 @@ +import { QueryClient, QueryClientProvider, useQuery } from '@tanstack/react-query'; +import { PathnameContext } from 'next/dist/shared/lib/hooks-client-context.shared-runtime'; +import { useState } from 'react'; + +import { TelemetryDisplayControls } from '@/components/gpu-power/TelemetryDisplayControls'; +import { + DEFAULT_TELEMETRY_DISPLAY, + type TelemetryDisplayState, +} from '@/components/gpu-power/telemetry-smoothing'; +import { PowerTelemetryView } from '@/components/inference/agentic-point/power-telemetry-view'; +import type { GpuMetricsPointPayload, GpuMetricSeries } from '@/hooks/api/use-gpu-metrics-point'; +import { registerAnalyticsClient } from '@/lib/analytics'; + +const ID = 206887; +const endpoint = `/api/v1/gpu-metrics-point?id=${ID}`; +const queryKey = ['gpu-metrics-point', ID]; +const chart = '[data-testid="power-telemetry-chart"]'; + +function series(id: number, host: string, base: number): GpuMetricSeries { + const data = [0, 1, 2].flatMap((second) => + [0, 1].map((index) => ({ + timestamp: `2026-09-21T00:00:0${second}Z`, + index, + power: base + index * 100 + second * 10, + temperature: 40 + index + second, + })), + ); + return { + id, + artifactName: 'gpu_metrics_qwen35_b200', + configKey: 'qwen35_b200_tp8', + fileName: `${host}/gpu_metrics.csv`, + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: data.length, + gpuCount: 2, + startedAt: data[0].timestamp, + endedAt: data.at(-1)!.timestamp, + sidecars: {}, + benchmarkResultIds: [ID], + stats: [0, 1].map((gpuIndex) => ({ + gpuIndex, + metric: 'power_w', + count: 3, + min: base + gpuIndex * 100, + max: base + gpuIndex * 100 + 20, + mean: base + gpuIndex * 100 + 10, + median: base + gpuIndex * 100 + 10, + p95: base + gpuIndex * 100 + 19, + p99: base + gpuIndex * 100 + 19.8, + stddev: Math.sqrt(200 / 3), + })), + data, + }; +} + +const payload: GpuMetricsPointPayload = { + benchmarkResultId: ID, + series: [series(1, 'host-a', 500), series(2, 'host-b', 700)], +}; + +function QueryStatus() { + const { status } = useQuery({ queryKey, enabled: false }); + return {status}; +} + +function mountPoint(path = '/inference/agentic/206887') { + const client = new QueryClient({ defaultOptions: { queries: { retry: false, gcTime: 0 } } }); + cy.mount( + + +
    + + +
    +
    +
    , + ); + return client; +} + +function Controls() { + const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); + return ( + <> + + {JSON.stringify(display)} + + ); +} + +describe('PowerX telemetry interactions', () => { + beforeEach(() => { + registerAnalyticsClient({ capture: cy.stub().as('capture') }); + }); + + it('changes display, window and chip aggregation independently and tracks each action', () => { + cy.mount(); + cy.get('[data-testid="display-mode-points"]').click(); + cy.get('#display-window').should('not.exist'); + cy.get('[data-testid="display-series-mean"]').click(); + cy.get('[data-testid="display-mode-rolling"]').click(); + cy.get('#display-window').click(); + cy.get('[role="option"]').contains('60 s').click(); + cy.get('output').should('have.text', '{"mode":"rolling","windowS":60,"series":"mean"}'); + cy.get('@capture').should('have.been.calledWith', 'power_test_display_mode_changed', { + mode: 'points', + }); + cy.get('@capture').should('have.been.calledWith', 'power_test_series_mode_changed', { + series: 'mean', + }); + cy.get('@capture').should('have.been.calledWith', 'power_test_smoothing_window_changed', { + windowS: 60, + }); + cy.get('@capture').its('callCount').should('eq', 4); + }); + + it('shows loading, distinguishes an initial failure from missing data, and retries', () => { + let failed = true; + cy.intercept('GET', endpoint, (request) => { + request.reply(failed ? { statusCode: 503, delay: 150, body: {} } : { body: payload }); + }).as('telemetry'); + mountPoint(); + cy.get('[data-testid="power-telemetry-loading"]').should('contain.text', 'Loading PowerX'); + cy.wait('@telemetry'); + cy.get('[data-testid="power-telemetry-query-error"]').should('contain.text', 'Failed to load'); + cy.get('[data-testid="power-telemetry-missing"]').should('not.exist'); + cy.then(() => { + failed = false; + }); + cy.contains('button', 'Retry').click(); + cy.wait('@telemetry'); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.get('@capture').should( + 'have.been.calledWith', + 'inference_agentic_power_telemetry_retry_clicked', + ); + }); + + it('shows a genuine missing point without offering an error retry', () => { + cy.intercept('GET', endpoint, { statusCode: 404, body: {} }); + mountPoint(); + cy.get('[data-testid="power-telemetry-missing"]').should('contain.text', `#${ID}`); + cy.get('[data-testid="power-telemetry-query-error"]').should('not.exist'); + cy.get(chart).should('not.exist'); + }); + + it('preserves a rendered chart when a background refetch fails', () => { + let failed = false; + cy.intercept('GET', endpoint, (request) => { + request.reply(failed ? { statusCode: 503, body: {} } : { body: payload }); + }).as('telemetry'); + const client = mountPoint(); + cy.wait('@telemetry'); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.then(() => { + failed = true; + return client.refetchQueries({ queryKey }); + }); + cy.wait('@telemetry'); + cy.get('[data-testid="query-status"]').should('have.text', 'error'); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.get('[data-testid="power-telemetry-query-error"]').should('not.exist'); + }); + + for (const [locale, width] of [ + ['en', 1280], + ['zh', 390], + ] as const) { + it(`keeps chip filters scoped to a host and renders ${locale} at ${width}px`, () => { + cy.viewport(width, 900); + cy.intercept('GET', endpoint, { body: payload }); + mountPoint(`${locale === 'zh' ? '/zh' : ''}/inference/agentic/${ID}`); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.get('[data-testid="power-telemetry-sample-count"]').should('have.text', '6'); + cy.get('table tbody tr').first().find('td').eq(4).should('have.text', '510.0'); + cy.get('[data-testid="chart-legend"]') + .contains(locale === 'zh' ? '芯片 0' : 'Chip 0') + .click(); + cy.get(chart).find('svg .point').should('have.length', 3); + cy.get('#power-telemetry-series-select').click(); + cy.get('[role="option"]').contains('host-b').click(); + cy.get(chart).find('svg .point').should('have.length', 6); + cy.get('table tbody tr').first().find('td').eq(4).should('have.text', '710.0'); + cy.get('[data-testid="power-telemetry-display-mode-points"]').click(); + cy.get('[data-testid="power-telemetry-display-series-mean"]').click(); + cy.get(chart).find('svg .point').should('have.length', 3); + cy.get('[data-testid="power-telemetry-view"]').should(($view) => { + expect($view[0].scrollWidth).to.be.at.most($view[0].clientWidth + 1); + }); + cy.get('[data-testid="power-telemetry-metric-select"]').should( + 'contain.text', + locale === 'zh' ? '功耗' : 'Power', + ); + cy.get('[data-testid="power-telemetry-view"]').screenshot( + `power-telemetry-${locale}-${width}`, + ); + }); + } +}); diff --git a/packages/app/src/components/inference/agentic-point/agentic-point-detail.tsx b/packages/app/src/components/inference/agentic-point/agentic-point-detail.tsx index 369078c40..f53d51dce 100644 --- a/packages/app/src/components/inference/agentic-point/agentic-point-detail.tsx +++ b/packages/app/src/components/inference/agentic-point/agentic-point-detail.tsx @@ -33,6 +33,7 @@ import { type StagePhase, } from './phase-slice'; import { PointSummary } from './point-summary'; +import { PowerTelemetryView } from './power-telemetry-view'; import { RequestMetricOverTime, SequenceMetricCard } from './request-metric-cards'; import { ServerLogViewer } from './server-log-viewer'; import { @@ -61,6 +62,7 @@ export const AGENTIC_POINT_DETAIL_STRINGS = { ttftOverTime: 'TTFT over time', perPoint: 'Per-point', requestTimeline: 'Request timeline', + powerX: 'PowerX', aggregatesAcrossConfigs: 'Aggregates across configs', logs: 'Logs', detailView: 'Detail view', @@ -94,6 +96,7 @@ export const AGENTIC_POINT_DETAIL_STRINGS = { ttftOverTime: 'TTFT 随时间变化', perPoint: '单点', requestTimeline: '请求时间线', + powerX: 'PowerX', aggregatesAcrossConfigs: '跨配置聚合', logs: '日志', detailView: '详情视图', @@ -149,6 +152,7 @@ export function AgenticPointDetail({ id }: Props) { () => [ { value: 'point', label: t.perPoint, testId: 'detail-view-point' }, { value: 'timeline', label: t.requestTimeline, testId: 'detail-view-timeline' }, + { value: 'power', label: t.powerX, testId: 'detail-view-power' }, { value: 'aggregates', label: t.aggregatesAcrossConfigs, testId: 'detail-view-aggregates' }, { value: 'logs', label: t.logs, testId: 'detail-view-logs' }, ], @@ -347,6 +351,8 @@ export function AgenticPointDetail({ id }: Props) { {view === 'logs' ? ( + ) : view === 'power' ? ( + ) : view === 'aggregates' ? ( aggregatesQuery.isError ? ( { t: number; value: number }[]; +} + +const percent = (points: readonly { t: number; value: number }[]) => + points.map((p) => ({ t: p.t, value: p.value * 100 })); + +/** + * Menu of overlay candidates, in display order. Only sources whose series is + * non-empty for the point are offered (see `availableOverlaySources`). + */ +export const OVERLAY_SOURCES: readonly OverlaySource[] = [ + { + key: 'decodeTps', + label: { en: 'Decode throughput', zh: 'Decode 吞吐量' }, + unit: 'tok/s', + color: '#8b5cf6', + points: (m) => m.decodeTps, + }, + { + key: 'prefillTps', + label: { en: 'Prefill throughput', zh: 'Prefill 吞吐量' }, + unit: 'tok/s', + color: '#06b6d4', + points: (m) => m.prefillTps, + }, + { + key: 'kvCacheUsage', + label: { en: 'KV cache utilization', zh: 'KV cache 利用率' }, + unit: '%', + color: '#f59e0b', + points: (m) => percent(m.kvCacheUsage), + }, + { + key: 'hostKvCacheUsage', + label: { en: 'Host KV cache utilization', zh: '主机 KV cache 利用率' }, + unit: '%', + color: '#d97706', + points: (m) => percent(m.hostKvCacheUsage), + }, + { + key: 'prefixCacheHitRate', + label: { en: 'Prefix cache hit rate', zh: 'Prefix cache 命中率' }, + unit: '%', + color: '#10b981', + points: (m) => percent(m.prefixCacheHitRate), + }, + { + key: 'prefixCacheHitsTps', + label: { en: 'Prefix cache hits', zh: 'Prefix cache 命中量' }, + unit: 'tok/s', + color: '#14b8a6', + points: (m) => m.prefixCacheHitsTps, + }, + { + key: 'queueDepth', + label: { en: 'Queue depth (running + waiting)', zh: '队列深度(运行中 + 等待中)' }, + unit: 'req', + color: '#ec4899', + points: (m) => m.queueDepth.map((p) => ({ t: p.t, value: p.total })), + }, +]; + +/** Sources that have at least one sample for this point, in menu order. */ +export function availableOverlaySources( + metrics: TraceServerMetrics | null | undefined, +): OverlaySource[] { + if (!metrics) return []; + return OVERLAY_SOURCES.filter((source) => source.points(metrics).length > 0); +} + +export function overlaySourceLabel(source: OverlaySource, locale: Locale): string { + return source.label[locale]; +} diff --git a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx new file mode 100644 index 000000000..7d56ef9a2 --- /dev/null +++ b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx @@ -0,0 +1,448 @@ +'use client'; + +import { useMemo, useState } from 'react'; + +import GpuMetricsChart, { + GPU_COLORS, + type TelemetryOverlaySeries, +} from '@/components/gpu-power/GpuPowerChart'; +import GpuStatsTable from '@/components/gpu-power/GpuStatsTable'; +import { TelemetryDisplayControls } from '@/components/gpu-power/TelemetryDisplayControls'; +import { + DEFAULT_TELEMETRY_DISPLAY, + toAbsoluteMs, + type TelemetryDisplayState, +} from '@/components/gpu-power/telemetry-smoothing'; +import { + type GpuMetricKey, + type GpuMetricRow, + ALL_METRIC_OPTIONS, + getAvailableMetrics, + getGpuMetricLabel, +} from '@/components/gpu-power/types'; +import { Card } from '@/components/ui/card'; +import ChartLegend from '@/components/ui/chart-legend'; +import { Label } from '@/components/ui/label'; +import { RetryableQueryError } from '@/components/ui/retryable-query-error'; +import { + Select, + SelectContent, + SelectItem, + SelectTrigger, + SelectValue, +} from '@/components/ui/select'; +import { useGpuMetricsPoint, type GpuMetricSeries } from '@/hooks/api/use-gpu-metrics-point'; +import { useTraceServerMetrics } from '@/hooks/api/use-trace-server-metrics'; + +import { availableOverlaySources, overlaySourceLabel } from './overlay-sources'; +import { track } from '@/lib/analytics'; +import { useLocale } from '@/lib/use-locale'; + +const STRINGS = { + en: { + loading: 'Loading PowerX telemetry…', + error: 'Failed to load PowerX telemetry.', + missing: + 'No PowerX telemetry is stored for benchmark point #{id}. Telemetry may not have been collected or ingested.', + series: 'Telemetry series', + metric: 'Metric', + vendor: 'Collector', + samples: 'Samples', + chips: 'Chips', + interval: 'Sample interval', + window: 'Recorded window', + sharedNote: + 'This series covers the whole benchmark job, including server start-up and warm-up, so summary rows below span more than the measured serving window.', + perGpuStats: 'Per-chip statistics', + chip: 'Chip', + secondsUnit: 's', + resetFilter: 'Show all chips', + overlayToggle: 'Overlay server metric', + overlayNone: 'None', + overlayLoading: 'Loading server metrics…', + overlayError: 'Server metrics failed to load; overlays are unavailable.', + overlayUnavailable: 'This point has no server-metric series to overlay.', + overlayAligned: 'The overlay is aligned to the telemetry by wall-clock timestamps.', + overlayRelative: + 'The trace has no wall-clock timestamps, so the overlay and the telemetry are both aligned at their own t=0.', + }, + zh: { + loading: '正在加载 PowerX 遥测数据……', + error: 'PowerX 遥测数据加载失败。', + missing: '基准测试数据点 #{id} 没有存储的 PowerX 遥测数据。遥测数据可能尚未采集或入库。', + series: '遥测序列', + metric: '指标', + vendor: '采集器', + samples: '样本数', + chips: '芯片数', + interval: '采样间隔', + window: '记录时间窗口', + sharedNote: + '该序列覆盖整个基准测试任务,包括服务启动与 warmup 阶段,因此下方统计范围大于实际测量的服务窗口。', + perGpuStats: '单芯片统计信息', + chip: '芯片', + secondsUnit: '秒', + resetFilter: '显示全部芯片', + overlayToggle: '叠加服务端指标', + overlayNone: '无', + overlayLoading: '正在加载服务端指标……', + overlayError: '服务端指标加载失败,无法叠加显示。', + overlayUnavailable: '该数据点没有可叠加的服务端指标序列。', + overlayAligned: '叠加曲线已按绝对时间戳与遥测数据对齐。', + overlayRelative: 'trace 缺少绝对时间戳,因此叠加曲线与遥测数据均从各自的 t=0 开始对齐。', + }, +} as const; + +const VENDOR_LABEL: Record = { nvidia: 'nvidia-smi', amd: 'amd-smi' }; + +/** + * Single-node CSVs come from the vendor CLI; multinode power bundles record + * their own producer (e.g. `srt-slurm.dcgm-power`) in the context sidecar. + */ +export function collectorLabel(series: Pick): string { + const context = series.sidecars?.context; + const producer = + context && typeof context === 'object' ? (context as { producer?: unknown }).producer : null; + if (typeof producer === 'string' && producer.trim() !== '') return producer; + return VENDOR_LABEL[series.vendor] ?? series.vendor; +} + +interface Props { + id: number; + enabled: boolean; + /** The point's hardware key, for the TDP reference line. */ + hardware?: string; + /** Fixed-sequence points do not have AgentX server-metric overlays. */ + serverMetricsEnabled?: boolean; +} + +function seriesLabel(series: GpuMetricSeries, total: number): string { + return total > 1 ? `${series.artifactName} · ${series.fileName}` : series.artifactName; +} + +/** + * PowerX tab of the per-point detail page: the full-resolution chip telemetry + * recorded while this benchmark point ran, read from the ingest-time digest + * (migration 016) rather than from GitHub artifacts. + */ +export function PowerTelemetryView({ id, enabled, hardware, serverMetricsEnabled = true }: Props) { + const locale = useLocale(); + const t = STRINGS[locale]; + const query = useGpuMetricsPoint(id, enabled); + const seriesList = query.data?.series ?? []; + + const [seriesSelection, setSeriesSelection] = useState<{ id: number; seriesId: number } | null>( + null, + ); + const selectedSeries = + (seriesSelection?.id === id + ? seriesList.find((series) => series.id === seriesSelection.seriesId) + : undefined) ?? seriesList[0]; + const data: GpuMetricRow[] = useMemo(() => selectedSeries?.data ?? [], [selectedSeries]); + const availableMetrics = useMemo(() => getAvailableMetrics(data), [data]); + + const [metricSelection, setMetricSelection] = useState('power'); + const metricKey: GpuMetricKey = availableMetrics.some((m) => m.key === metricSelection) + ? metricSelection + : 'power'; + const metricConfig = ALL_METRIC_OPTIONS.find((m) => m.key === metricKey)!; + const allGpuIndices = useMemo( + () => [...new Set(data.map((row) => row.index))].toSorted((a, b) => a - b), + [data], + ); + // Hidden chips are scoped to the series they were hidden on so switching + // series never carries over a stale filter. + const [hiddenSelection, setHiddenSelection] = useState<{ + seriesId: number; + hidden: number[]; + } | null>(null); + const hiddenGpus = useMemo( + () => + new Set( + hiddenSelection && hiddenSelection.seriesId === selectedSeries?.id + ? hiddenSelection.hidden + : [], + ), + [hiddenSelection, selectedSeries?.id], + ); + const visibleGpus = useMemo( + () => new Set(allGpuIndices.filter((gpuIndex) => !hiddenGpus.has(gpuIndex))), + [allGpuIndices, hiddenGpus], + ); + const toggleGpu = (gpuIndex: number) => { + if (!selectedSeries) return; + track('inference_agentic_power_gpu_toggled', { id, gpuIndex }); + const next = new Set(hiddenGpus); + if (next.has(gpuIndex)) next.delete(gpuIndex); + else next.add(gpuIndex); + setHiddenSelection({ seriesId: selectedSeries.id, hidden: [...next] }); + }; + const [isLegendExpanded, setIsLegendExpanded] = useState(true); + const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); + + // Server-metric overlay. The series are fetched as soon as the tab opens so + // the menu can list exactly the metrics this point has; one source at a time. + const metricsQuery = useTraceServerMetrics(id, enabled && serverMetricsEnabled); + const serverMetrics = metricsQuery.data; + const overlaySources = useMemo(() => availableOverlaySources(serverMetrics), [serverMetrics]); + const [overlaySelection, setOverlaySelection] = useState<{ id: number; key: string } | null>( + null, + ); + const overlayKey = overlaySelection?.id === id ? overlaySelection.key : 'none'; + const overlaySource = overlaySources.find((source) => source.key === overlayKey) ?? null; + // Trace timeslices carry epoch-ns starts, so both series can share wall-clock + // time. A zero startNs means the trace only has relative time. + const overlayAbsolute = Boolean(serverMetrics && serverMetrics.startNs > 0); + const overlay = useMemo(() => { + if (!overlaySource || !serverMetrics || !selectedSeries) return null; + const originMs = overlayAbsolute + ? serverMetrics.startNs / 1e6 + : new Date(selectedSeries.startedAt).getTime(); + return { + label: overlaySourceLabel(overlaySource, locale), + unit: overlaySource.unit, + color: overlaySource.color, + points: toAbsoluteMs(overlaySource.points(serverMetrics), originMs), + }; + }, [overlaySource, serverMetrics, selectedSeries, overlayAbsolute, locale]); + const overlayNote = ((): string | null => { + if (metricsQuery.isLoading) return t.overlayLoading; + if (metricsQuery.isError) return t.overlayError; + if (overlaySources.length === 0) return t.overlayUnavailable; + if (!overlay) return null; + return overlayAbsolute ? t.overlayAligned : t.overlayRelative; + })(); + + if (!enabled) return null; + + if (query.isLoading) { + return ( +
    + {t.loading} +
    + ); + } + if (query.isError && !query.data) { + return ( + + ); + } + if (!selectedSeries) { + return ( +
    + {t.missing.replace('{id}', String(id))} +
    + ); + } + + const durationS = Math.max( + 0, + (new Date(selectedSeries.endedAt).getTime() - new Date(selectedSeries.startedAt).getTime()) / + 1000, + ); + const numberLocale = locale === 'zh' ? 'zh-CN' : undefined; + + return ( +
    + +
    +
    +
    {t.vendor}
    +
    {collectorLabel(selectedSeries)}
    +
    +
    +
    {t.samples}
    +
    + {selectedSeries.sampleCount.toLocaleString(numberLocale)} +
    +
    +
    +
    {t.chips}
    +
    {selectedSeries.gpuCount}
    +
    +
    +
    {t.interval}
    +
    + {selectedSeries.sampleIntervalS === null + ? '—' + : `${selectedSeries.sampleIntervalS.toFixed(2)} ${t.secondsUnit}`} +
    +
    +
    +
    {t.window}
    +
    + {new Date(selectedSeries.startedAt).toLocaleTimeString(numberLocale)} ·{' '} + {Math.round(durationS).toLocaleString(numberLocale)} {t.secondsUnit} +
    +
    +
    +
    + {seriesList.length > 1 && ( +
    + + +
    + )} +
    + + +
    +
    + + {serverMetricsEnabled && ( +
    +
    + + +
    + {overlayNote && ( + + {overlayNote} + + )} +
    + )} +
    + + + ({ + name: `${t.chip} ${gpuIndex}`, + hw: String(gpuIndex), + label: `${t.chip} ${gpuIndex}`, + color: GPU_COLORS[gpuIndex % GPU_COLORS.length], + isActive: visibleGpus.has(gpuIndex), + onClick: () => toggleGpu(gpuIndex), + }))} + onItemRemove={(hw) => { + const gpuIndex = Number(hw); + if (visibleGpus.has(gpuIndex)) toggleGpu(gpuIndex); + }} + isLegendExpanded={isLegendExpanded} + onExpandedChange={(expanded) => { + setIsLegendExpanded(expanded); + track('inference_agentic_power_legend_expanded', { id, expanded }); + }} + actions={ + hiddenGpus.size === 0 + ? [] + : [ + { + id: 'power-telemetry-show-all-chips', + label: t.resetFilter, + onClick: () => { + track('inference_agentic_power_gpu_reset_filter', { id }); + setHiddenSelection(null); + }, + }, + ] + } + /> + } + caption={ + + {getGpuMetricLabel(metricConfig, locale)} · {t.sharedNote} + + } + /> + + + +

    {t.perGpuStats}

    + +
    +
    + ); +} diff --git a/packages/app/src/components/inference/agentic-point/use-detail-view.ts b/packages/app/src/components/inference/agentic-point/use-detail-view.ts index 1a97a85cc..b08b3a868 100644 --- a/packages/app/src/components/inference/agentic-point/use-detail-view.ts +++ b/packages/app/src/components/inference/agentic-point/use-detail-view.ts @@ -6,10 +6,14 @@ import { useClientSearchParams } from '@/hooks/useClientSearch'; import { track } from '@/lib/analytics'; import { replaceClientSearch } from '@/lib/client-navigation'; -export type DetailView = 'point' | 'timeline' | 'aggregates' | 'logs'; +export type DetailView = 'point' | 'timeline' | 'power' | 'aggregates' | 'logs'; const isDetailView = (value: string | null): value is DetailView => - value === 'point' || value === 'timeline' || value === 'aggregates' || value === 'logs'; + value === 'point' || + value === 'timeline' || + value === 'power' || + value === 'aggregates' || + value === 'logs'; /** URL-persisted detail view (`?view=`; per-point is the unadorned default). */ export function useDetailView(): [DetailView, (nextView: DetailView) => void] { diff --git a/packages/app/src/components/inference/power-telemetry-dialog.tsx b/packages/app/src/components/inference/power-telemetry-dialog.tsx new file mode 100644 index 000000000..e20718a55 --- /dev/null +++ b/packages/app/src/components/inference/power-telemetry-dialog.tsx @@ -0,0 +1,51 @@ +'use client'; + +import type { InferenceData } from '@/components/inference/types'; +import { PowerTelemetryView } from '@/components/inference/agentic-point/power-telemetry-view'; +import { + Dialog, + DialogContent, + DialogDescription, + DialogHeader, + DialogTitle, +} from '@/components/ui/dialog'; +import { isPersistedBenchmarkId } from '@/lib/benchmark-id'; +import { useLocale } from '@/lib/use-locale'; + +const STRINGS = { + en: { point: 'Benchmark point', concurrency: 'Concurrency' }, + zh: { point: '基准测试数据点', concurrency: '并发数' }, +} as const; + +interface Props { + point: InferenceData; + onOpenChange: (open: boolean) => void; +} + +export function PowerTelemetryDialog({ point, onOpenChange }: Props) { + const t = STRINGS[useLocale()]; + if (!isPersistedBenchmarkId(point.id)) return null; + + return ( + + + + PowerX + + {t.point} #{point.id} · {point.hwKey} · {point.precision.toUpperCase()} ·{' '} + {t.concurrency} {point.conc} + + + + + + ); +} From 38b2d3ebb78e6f251ebcf76612ef4cdb2a6c276c Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Tue, 29 Sep 2026 12:27:46 -0700 Subject: [PATCH 4/7] feat(ui): measured power timeline MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Draw the one-second power traces of the drawn runs on a shared time axis, grouped by pool with reference labels, honoring hardware toggles and unofficial-run overlays. 中文:在共享时间轴上绘制已绘 run 的每秒功耗轨迹,按 pool 分组并带参考标签,遵循硬件开关与非正式 run 叠加。 --- .../cypress/component/power-timeline.cy.tsx | 184 ++ .../components/inference/ui/PowerTimeline.tsx | 1481 +++++++++++++++++ .../inference/utils/powerTimeline.test.ts | 21 + .../inference/utils/powerTimeline.ts | 344 ++++ 4 files changed, 2030 insertions(+) create mode 100644 packages/app/cypress/component/power-timeline.cy.tsx create mode 100644 packages/app/src/components/inference/ui/PowerTimeline.tsx create mode 100644 packages/app/src/components/inference/utils/powerTimeline.test.ts create mode 100644 packages/app/src/components/inference/utils/powerTimeline.ts diff --git a/packages/app/cypress/component/power-timeline.cy.tsx b/packages/app/cypress/component/power-timeline.cy.tsx new file mode 100644 index 000000000..01fc78f13 --- /dev/null +++ b/packages/app/cypress/component/power-timeline.cy.tsx @@ -0,0 +1,184 @@ +import { PathnameContext } from 'next/dist/shared/lib/hooks-client-context.shared-runtime'; + +import type { GpuPowerSeries, GpuPowerSeriesResponse } from '@/components/gpu-power/power-series'; +import PowerTimeline from '@/components/inference/ui/PowerTimeline'; +import type { InferenceData } from '@/components/inference/types'; +import { Model, Precision, Sequence } from '@/lib/data-mappings'; +import { overlayRunColor } from '@/lib/overlay-run-style'; + +import { + createMockHardwareConfig, + createMockInferenceData, + createMockUnofficialRunContext, +} from '../support/mock-data'; +import { mountWithProviders } from '../support/test-utils'; + +// PowerTimeline joins chart points to `gpu_metrics_` artifacts +// by the `power_audit.source` file name and draws one trace per config. Overlay +// runs keep their run colour and follow the overlay hardware filter. + +const RUN_ID = '34716669498'; +const RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${RUN_ID}`; +const OVERLAY_RUN_ID = '31415926535'; +const OVERLAY_RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${OVERLAY_RUN_ID}`; +const START_MS = Date.UTC(2026, 8, 12, 20, 20, 0); +const hwConfig = createMockHardwareConfig(); +const HW_TYPES = new Set(['b200', 'h100']); + +const resultName = (hardware: string, conc: number) => + `dsv4_8k1k_fp4_sglang_tp8-pp1-dcp1-pcp1-ep1-dpafalse_disagg-false_spec-none_conc${conc}_${hardware}-host-0123456789abcdef0123`; +const WINDOW = { + // Window covers the last 20 s of a 60 s job. + window_start_unix: (START_MS + 40_000) / 1000, + window_end_unix: (START_MS + 60_000) / 1000, +}; + +function measuredPoint( + hwKey: string, + conc: number, + watts: number, + overrides: Partial = {}, +): InferenceData { + return createMockInferenceData({ + hwKey, + conc, + tp: 8, + x: conc, + y: watts, + precision: Precision.FP4, + run_url: RUN_URL, + measuredAvgPower: { y: watts, roof: true }, + measuredPowerTimeline: { y: watts, roof: true }, + power_audit: { + source: `power_validation_${resultName(hwKey.split('_')[0], conc)}.json`, + ...WINDOW, + }, + ...overrides, + }); +} + +/** 61 one-second buckets: idle 200 W, ramps to `peak` inside the window. */ +function series(hwKey: string, conc: number, peak: number, gpus = [0, 1]): GpuPowerSeries { + const t = Array.from({ length: 61 }, (_, i) => i); + return { + artifact: `gpu_metrics_${resultName(hwKey.split('_')[0], conc)}`, + startMs: START_MS, + bucketSeconds: 1, + gpus, + t, + power: gpus.map((gpu) => t.map((second) => (second >= 40 ? peak + gpu * 10 : 200 + gpu))), + }; +} + +const response: GpuPowerSeriesResponse = { + runInfo: { + id: Number(RUN_ID), + name: 'Run Sweep', + branch: 'main', + sha: 'abc123', + createdAt: '2026-09-12T20:00:00Z', + url: RUN_URL, + conclusion: 'success', + status: 'completed', + }, + series: [series('b200', 16, 700), series('b200', 64, 900)], +}; + +function mountTimeline( + data: InferenceData[], + options: { + overlay?: Parameters[0]['overlayData']; + unofficial?: Parameters[0]; + } = {}, +) { + mountWithProviders( + +
    + +
    +
    , + { + inference: { + selectedModel: Model.DeepSeek_V4_Pro, + selectedSequence: Sequence.EightK_OneK, + selectedYAxisMetric: 'y_measuredPowerTimeline', + hardwareConfig: hwConfig, + activeHwTypes: new Set(HW_TYPES), + hwTypesWithData: new Set(HW_TYPES), + }, + unofficial: options.unofficial ?? {}, + }, + ); +} + +const svg = () => cy.get('[data-testid="power-timeline-chart-svg"]'); + +describe('PowerTimeline', () => { + beforeEach(() => { + cy.on('uncaught:exception', (error) => { + if (error.message.includes('ResizeObserver loop')) return false; + }); + }); + + it('colours overlay-run traces by run and honours the overlay hardware filter', () => { + const overlayPoint = measuredPoint('h200', 16, 500, { + run_url: OVERLAY_RUN_URL, + power_audit: { + source: `power_validation_${resultName('h200', 16)}.json`, + ...WINDOW, + }, + }); + const overlayResponse: GpuPowerSeriesResponse = { + runInfo: { ...response.runInfo, id: Number(OVERLAY_RUN_ID), url: OVERLAY_RUN_URL }, + series: [series('h200', 16, 500)], + }; + cy.intercept('POST', `/api/gpu-metrics?runId=${RUN_ID}*`, { body: response }).as('official'); + cy.intercept('POST', `/api/gpu-metrics?runId=${OVERLAY_RUN_ID}*`, { + body: overlayResponse, + }).as('overlay'); + mountTimeline([measuredPoint('b200', 16, 700)], { + overlay: { + data: [overlayPoint], + hardwareConfig: hwConfig, + label: 'powerx-timeline', + runUrl: OVERLAY_RUN_URL, + }, + unofficial: createMockUnofficialRunContext({ + isUnofficialRun: true, + unofficialRunInfos: [ + { + id: Number(OVERLAY_RUN_ID), + name: 'powerx-timeline', + branch: 'powerx-timeline', + sha: 'abc000', + createdAt: '2026-09-12T00:00:00Z', + url: OVERLAY_RUN_URL, + conclusion: 'success', + status: 'completed', + isNonMainBranch: true, + }, + ], + runIndexByUrl: { [OVERLAY_RUN_URL]: 0, [OVERLAY_RUN_ID]: 0 }, + activeOverlayHwTypes: new Set(['h200']), + }), + }); + cy.wait(['@official', '@overlay']); + + svg().within(() => { + cy.get('path.power-trace[data-run-index="0"][data-segment="window"]') + .should('have.length', 1) + .and('have.attr', 'stroke', overlayRunColor(0)); + cy.get('path.power-trace[data-hw="b200"][data-segment="window"]').should('have.length', 1); + // Two runs: the axis defaults to elapsed time so traces overlap by phase. + cy.get('text').contains('Time since telemetry start').should('exist'); + // Reference lines cover both hardware SKUs. + cy.get('.power-reference[data-reference="tdp"]').should('have.length', 2); + }); + cy.get('[data-testid="chart-legend"]').should('contain.text', '✕ powerx-timeline'); + }); +}); diff --git a/packages/app/src/components/inference/ui/PowerTimeline.tsx b/packages/app/src/components/inference/ui/PowerTimeline.tsx new file mode 100644 index 000000000..6a56f0c4d --- /dev/null +++ b/packages/app/src/components/inference/ui/PowerTimeline.tsx @@ -0,0 +1,1481 @@ +'use client'; + +/** + * PowerX "Timeline" display: the per-second GPU power behind each measured + * average, drawn over the whole benchmark job. + * + * ChartDisplay renders this instead of ScatterGraph when the Measured Power + * Display control is `timeline` (`y_measuredPowerTimeline`). The point set is + * the same as the Measured Avg Power axis; each point's `power_audit.source` + * names its `gpu_metrics_*` artifact — or, for disaggregated Slurm / Dynamo + * rows, its validation file inside a `power_audit_*` bundle — which + * `/api/gpu-metrics?series=power` returns as one-second per-GPU buckets + * (`components/gpu-power/power-series.ts`). + * + * One trace per config, coloured by hardware (official) or by run (unofficial + * overlay). The validated measurement window is emphasized; the rest of the + * job (server start, warmup) is drawn faint. Rated TDP is a dashed reference + * per hardware; the all-in provisioned line is opt-in because it would halve + * the vertical resolution of the traces. + * + * Pool mode (bundle series carry worker roles) sums each prefill / decode pool + * instead, against pool-sized TDP references, so a disaggregated deployment + * reads as two lines on a watts axis. A pinned scatter tooltip can deep-link + * here focused on one config (`requestPowerTraceFocus`). + */ +import * as d3 from 'd3'; +import { useQueries } from '@tanstack/react-query'; +import React, { useCallback, useEffect, useMemo, useRef, useState } from 'react'; +import { HW_REGISTRY } from '@semianalysisai/inferencex-constants'; + +import { + bucketTimeMs, + meanPowerAt, + sumPowerAt, + type GpuPowerSeries, + type GpuPowerSeriesResponse, +} from '@/components/gpu-power/power-series'; +import ChartLegend, { type LegendSwitchConfig } from '@/components/ui/chart-legend'; +import { SegmentedToggle } from '@/components/ui/segmented-toggle'; +import { useUnofficialRun } from '@/components/unofficial-run-provider'; +import { matchesQuickFilters } from '@/components/inference/utils/quickFilters'; +import { useThemeColors } from '@/hooks/useThemeColors'; +import { track } from '@/lib/analytics'; +import { computeToggle } from '@/lib/toggle-set'; +import { getModelSortIndex } from '@/lib/constants'; +import { D3Chart } from '@/lib/d3-chart/D3Chart'; +import type { LayerConfig, RenderContext, ZoomContext } from '@/lib/d3-chart/D3Chart/types'; +import { lttbDownsample } from '@/lib/d3-chart/downsample'; +import { CHART_FONT_SANS, CHART_TYPE, px } from '@/lib/d3-chart/typography'; +import { overlayRunColor, overlayRunIndex } from '@/lib/overlay-run-style'; +import { useLocale } from '@/lib/use-locale'; +import { getDisplayLabel } from '@/lib/utils'; + +import { + useInferenceActions, + useInferenceData, + useInferenceDisplay, + useInferenceFilters, +} from '../InferenceContext'; +import type { InferenceData, OverlayData } from '../types'; +import { powerVariantDash } from '../utils/power-compare'; +import { + allGpuPool, + consumePowerTraceFocus, + groupPoolsBySize, + joinPowerTimeline, + planPowerTimelineRequests, + prioritizeRun, + prioritizeRuns, + referenceLabelSlots, + runIdFromUrl, + traceConfigLabel, + traceKeyRunId, + tracePools, + windowPhase, + type MissingTrace, + type MissingTraceReason, + type PoolSizeGroup, + type PowerPool, + type PowerPoolRole, + type PowerTimelineRequest, + type PowerTimelineTrace, + type WindowPhase, +} from '../utils/powerTimeline'; + +/** Distinct workflow runs fetched per chart; each is one GitHub artifact sweep. */ +export const POWER_TIMELINE_MAX_RUNS = 4; +/** Hover targets per trace after LTTB; paths keep every bucket. */ +const HIT_POINTS_PER_TRACE = 200; +/** Above this many visible traces the `c` end labels would only overlap. */ +const MAX_LABELED_TRACES = 40; +/** Up to this many undrawn configs are named individually; beyond, per hardware. */ +const MAX_LISTED_MISSING = 8; +const CHART_HEIGHT = 600; +const MARGIN = { top: 24, right: 84, bottom: 60, left: 64 }; + +const STRINGS = { + en: { + timeAxis: 'Time axis', + wall: 'Wall clock (UTC)', + elapsed: 'Since start', + xWall: 'Time (UTC)', + xElapsed: 'Time since telemetry start (m:ss)', + perGpu: 'One line per GPU', + perGpuHelp: 'Draw every GPU of a config instead of the mean across its GPUs.', + pools: 'Prefill / decode pools', + poolsHelp: + 'One line per worker-role pool: the summed board power of the prefill GPUs and of the decode GPUs of a config. Dashed references are pool size × rated TDP.', + utilityLines: 'All-in provisioned lines', + utilityHelp: + 'Dashed reference at the all-in provisioned utility power per GPU from the hardware registry (SemiAnalysis Datacenter Industry Model). Off by default because it compresses the traces.', + loading: (runs: number) => + `Loading GPU telemetry for ${runs} run${runs === 1 ? '' : 's'}… (may take a minute)`, + loadError: (runId: string, message: string) => `Run ${runId}: ${message}`, + missing: (missing: number, total: number) => + `${missing} of ${total} measured configs have no telemetry trace and are not drawn.`, + missingReason: { + 'no-source': (count: number) => + `${count} predate per-config telemetry provenance in the benchmark row`, + 'no-run': (count: number) => `${count} carry no workflow run`, + 'run-not-fetched': (count: number) => `${count} come from runs that were not loaded`, + 'not-in-run': (count: number) => + `${count} have no gpu_metrics artifact or power-audit bundle in their run (expired, or another collector)`, + } satisfies Record string>, + missingUndrawn: 'Not drawn', + noTraces: + 'No telemetry traces for the visible hardware. Enable a series in the legend or choose another date.', + noArtifacts: + 'These points predate per-config telemetry artifacts, so no timeline is available for them.', + droppedRuns: (runs: number) => + `Telemetry from ${runs} more run${runs === 1 ? '' : 's'} was not loaded (limit ${POWER_TIMELINE_MAX_RUNS} runs per chart).`, + telemetry: 'Telemetry', + method: + 'One-second means of per-GPU board power (nvidia-smi / amd-smi, or DCGM on Slurm / Dynamo runs) over the whole benchmark job; the emphasized segment is the validated window behind the measured average. Dashed lines: rated TDP per hardware from the hardware registry.', + methodPools: + 'In pool mode each line is the summed power of one worker-role pool (prefill or decode GPUs) and the dashed references are pool size × rated TDP.', + instructions: + 'Shift+Scroll to zoom horizontally · Drag to pan · Double-click to reset · Click a point to pin tooltip', + dismiss: 'Click elsewhere to dismiss', + phase: { + before: 'Before window (startup / warmup)', + window: 'Measurement window', + after: 'After window', + unknown: 'Window not recorded', + } satisfies Record, + meanPerGpu: 'Mean per GPU', + gpus: (count: number) => `${count} GPU${count === 1 ? '' : 's'}`, + min: 'min', + max: 'max', + validated: 'Validated average', + sinceStart: 'since start', + tdp: 'TDP', + allIn: 'all-in', + poolShort: { prefill: 'prefill', decode: 'decode', all: 'all GPUs' } satisfies Record< + PowerPoolRole, + string + >, + yPool: 'GPU pool power (W)', + pool: 'Pool', + poolPower: 'Pool power', + poolTdp: 'pool TDP', + focused: (label: string) => `Focused on ${label}`, + showAll: 'Show all', + unofficialRun: 'Unofficial run', + branch: 'Branch', + viewWorkflow: 'View workflow run', + }, + zh: { + timeAxis: '时间轴', + wall: '实际时刻(UTC)', + elapsed: '相对起点', + xWall: '时间(UTC)', + xElapsed: '距遥测开始的时间(分:秒)', + perGpu: '每个 GPU 一条线', + perGpuHelp: '绘制配置中每个 GPU 的曲线,而不是各 GPU 的平均值。', + pools: '预填充 / 解码 GPU 池', + poolsHelp: + '按 worker 角色分池绘制:每条线是同一配置中预填充 GPU 或解码 GPU 的板卡功耗之和。虚线参考为池内 GPU 数量 × 额定 TDP。', + utilityLines: '全电源配置参考线', + utilityHelp: + '按硬件注册表中每 GPU 的全电源配置(all-in)市电功率绘制虚线参考(SemiAnalysis 数据中心行业模型)。默认关闭,因为它会压缩曲线的纵向分辨率。', + loading: (runs: number) => `正在加载 ${runs} 个运行的 GPU 遥测数据……(可能需要约一分钟)`, + loadError: (runId: string, message: string) => `运行 ${runId}:${message}`, + missing: (missing: number, total: number) => + `${total} 个有实测值的配置中有 ${missing} 个没有遥测曲线,未绘制。`, + missingReason: { + 'no-source': (count: number) => `${count} 个的基准测试行早于按配置记录的遥测来源`, + 'no-run': (count: number) => `${count} 个没有工作流运行信息`, + 'run-not-fetched': (count: number) => `${count} 个来自未加载的运行`, + 'not-in-run': (count: number) => + `${count} 个在其运行中没有 gpu_metrics 产物或 power-audit 数据包(产物已过期,或使用其他采集器)`, + } satisfies Record string>, + missingUndrawn: '未绘制', + noTraces: '当前可见硬件没有遥测曲线。请在图例中启用一个系列或选择其他日期。', + noArtifacts: '这些数据点早于按配置上传的遥测产物,因此没有可用的时间线。', + droppedRuns: (runs: number) => + `另有 ${runs} 个运行的遥测数据未加载(每张图表最多 ${POWER_TIMELINE_MAX_RUNS} 个运行)。`, + telemetry: '遥测来源', + method: + '整个基准测试任务期间每个 GPU 板卡功耗(nvidia-smi / amd-smi,Slurm / Dynamo 运行为 DCGM)的一秒平均值;加粗段为实测平均值所依据的有效测量窗口。虚线:硬件注册表中各硬件的额定 TDP。', + methodPools: + '在 GPU 池模式下,每条线是一个 worker 角色池(预填充或解码 GPU)的功耗总和,虚线参考为池内 GPU 数量 × 额定 TDP。', + instructions: 'Shift+滚轮横向缩放 · 拖动平移 · 双击重置 · 点击数据点固定提示框', + dismiss: '点击其他区域关闭', + phase: { + before: '测量窗口之前(启动 / warmup)', + window: '测量窗口内', + after: '测量窗口之后', + unknown: '未记录测量窗口', + } satisfies Record, + meanPerGpu: '每 GPU 平均', + gpus: (count: number) => `${count} 个 GPU`, + min: '最小', + max: '最大', + validated: '有效平均值', + sinceStart: '距起点', + tdp: 'TDP', + allIn: 'all-in', + poolShort: { prefill: '预填充', decode: '解码', all: '全部 GPU' } satisfies Record< + PowerPoolRole, + string + >, + yPool: 'GPU 池功耗(W)', + pool: 'GPU 池', + poolPower: '池功耗', + poolTdp: '池 TDP', + focused: (label: string) => `聚焦:${label}`, + showAll: '显示全部', + unofficialRun: '非官方运行', + branch: '分支', + viewWorkflow: '查看工作流运行', + }, +} as const; + +type XMode = 'wall' | 'elapsed'; +type LineMode = 'mean' | 'gpu' | 'pool'; + +interface TimelineSample { + trace: PowerTimelineTrace; + color: string; + overlayIndex: number | null; + column: number; + timeMs: number; + /** Data-space x for the active mode: epoch ms (wall) or seconds (elapsed). */ + x: number; + /** Mean watts across the GPUs sampled in the bucket; the pool's summed watts in pool mode. */ + y: number; + min: number; + max: number; + /** GPUs with a sample in the bucket (inside the pool, in pool mode). */ + gpuCount: number; + phase: WindowPhase; + /** The pool this sample sums and its device count, in pool mode. */ + pool?: { role: PowerPoolRole; gpuCount: number }; +} + +interface TracePoint { + x: number; + y: number | null; +} + +interface TracePath { + id: string; + traceKey: string; + hwKey: string; + overlayIndex: number | null; + color: string; + segment: 'full' | 'window'; + width: number; + opacity: number; + points: TracePoint[]; + /** Worker-role pool the line sums, in pool mode. */ + pool?: PowerPoolRole; +} + +interface ReferenceLine { + id: string; + watts: number; + label: string; + color: string; + kind: 'tdp' | 'utility'; + /** Pools the line is sized for, in pool mode (roles sharing one GPU count). */ + pools?: PowerPoolRole[]; +} + +/** Vertical distance between stacked reference labels that share a watts value. */ +const REFERENCE_LABEL_ROW = 13; + +interface TraceLabel { + /** Join key: the trace key, plus the pool role in pool mode. */ + id: string; + traceKey: string; + hwKey: string; + pool?: PowerPoolRole; + color: string; + text: string; + x: number; + y: number; +} + +interface DrawModel { + paths: TracePath[]; + labels: TraceLabel[]; +} + +export interface PowerTimelineProps { + chartId: string; + /** Official points of the chart (display-limit clipped points restored). */ + data: InferenceData[]; + overlayData?: OverlayData; + yLabel: string; + caption?: React.ReactNode; +} + +async function fetchPowerSeries( + request: PowerTimelineRequest, + signal: AbortSignal, +): Promise { + const params = new URLSearchParams({ runId: request.runId, series: 'power' }); + if (request.prefix) params.set('prefix', request.prefix); + const response = await fetch(`/api/gpu-metrics?${params.toString()}`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ sources: request.sources }), + cache: 'no-store', + signal, + }); + const body = (await response.json()) as GpuPowerSeriesResponse | { error: string }; + if (!response.ok) { + throw new Error('error' in body ? body.error : `HTTP ${response.status}`); + } + return body as GpuPowerSeriesResponse; +} + +function formatElapsed(totalSeconds: number): string { + const seconds = Math.max(0, Math.round(totalSeconds)); + const h = Math.floor(seconds / 3600); + const m = Math.floor((seconds % 3600) / 60); + const s = seconds % 60; + const mm = h > 0 ? String(m).padStart(2, '0') : String(m); + return `${h > 0 ? `${h}:` : ''}${mm}:${String(s).padStart(2, '0')}`; +} + +const formatUtcClock = d3.utcFormat('%H:%M:%S'); +const formatUtcDate = d3.utcFormat('%Y-%m-%d'); +/** Pool sums run to thousands of watts; group the digits. */ +const formatWatts = d3.format(',.0f'); + +function baseHardware(hwKey: string): string { + return hwKey.split('_')[0]; +} + +/** + * The pools a trace draws in pool mode: its worker-role pools, or every GPU as + * one pool when the collector assigned no roles (a single-node trace then shows + * its deployment total on the same axis). + */ +function drawnPools(series: GpuPowerSeries): PowerPool[] { + const pools = tracePools(series); + return pools.length > 0 ? pools : [allGpuPool(series)]; +} + +/** SVG dash of a pool line: per role from the comparison palette; `all` stays solid. */ +function poolDash(pool: PowerPoolRole): string | null { + const dash = powerVariantDash({ kind: 'role', id: pool }); + return dash === '' ? null : dash; +} + +interface TraceRow { + id: string; + pool?: PowerPoolRole; + values: (number | null)[]; +} + +/** One polyline's values per line mode: the GPU mean, each GPU, or each pool's sum. */ +function traceRows(series: GpuPowerSeries, lineMode: LineMode): TraceRow[] { + if (lineMode === 'gpu') { + return series.gpus.map((gpu, row) => ({ id: `gpu${gpu}`, values: series.power[row] })); + } + if (lineMode === 'pool') { + return drawnPools(series).map((pool) => ({ + id: `pool:${pool.role}`, + pool: pool.role, + values: series.t.map((_, column) => sumPowerAt(series, pool.rows, column)), + })); + } + return [{ id: 'mean', values: series.t.map((_, column) => meanPowerAt(series, column)) }]; +} + +/** Builds the mean, per-GPU or per-pool polylines plus the window emphasis for one trace. */ +function tracePaths( + trace: PowerTimelineTrace, + color: string, + overlayIndex: number | null, + xMode: XMode, + lineMode: LineMode, +): TracePath[] { + const { series } = trace; + const xOf = (column: number) => + xMode === 'wall' ? bucketTimeMs(series, column) : series.t[column] - series.t[0]; + const rows = traceRows(series, lineMode); + const faint = lineMode === 'gpu' ? 0.22 : 0.32; + const strong = lineMode === 'gpu' ? 0.85 : 1; + const widths = lineMode === 'gpu' ? [1, 1.5] : [1.25, 2.25]; + const paths: TracePath[] = []; + for (const row of rows) { + const full: TracePoint[] = []; + const window: TracePoint[] = []; + row.values.forEach((value, column) => { + const point = { x: xOf(column), y: value }; + full.push(point); + if (windowPhase(trace, bucketTimeMs(series, column)) === 'window') window.push(point); + }); + paths.push({ + id: `${trace.key}:${row.id}:full`, + traceKey: trace.key, + hwKey: trace.point.hwKey, + overlayIndex, + color, + segment: 'full', + width: widths[0], + opacity: faint, + points: full, + pool: row.pool, + }); + if (window.length > 1) { + paths.push({ + id: `${trace.key}:${row.id}:window`, + traceKey: trace.key, + hwKey: trace.point.hwKey, + overlayIndex, + color, + segment: 'window', + width: widths[1], + opacity: strong, + points: window, + pool: row.pool, + }); + } + } + return paths; +} + +/** + * Hover targets of one trace: one stream over all its GPUs (mean watts), or in + * pool mode one stream per pool (summed watts) so the tooltip can name the pool. + */ +function traceSamples( + trace: PowerTimelineTrace, + color: string, + overlayIndex: number | null, + xMode: XMode, + lineMode: LineMode, +): TimelineSample[] { + const { series } = trace; + if (lineMode !== 'pool') { + const rows = series.power.map((_, row) => row); + return sampleRows(trace, color, overlayIndex, xMode, rows, undefined); + } + return drawnPools(series).flatMap((pool) => + sampleRows(trace, color, overlayIndex, xMode, pool.rows, { + role: pool.role, + gpuCount: pool.rows.length, + }), + ); +} + +function sampleRows( + trace: PowerTimelineTrace, + color: string, + overlayIndex: number | null, + xMode: XMode, + rows: readonly number[], + pool: TimelineSample['pool'], +): TimelineSample[] { + const { series } = trace; + const samples: TimelineSample[] = []; + for (let column = 0; column < series.t.length; column++) { + let sum = 0; + let count = 0; + let min = Number.POSITIVE_INFINITY; + let max = Number.NEGATIVE_INFINITY; + for (const row of rows) { + const value = series.power[row]?.[column]; + if (value === null || value === undefined) continue; + sum += value; + count += 1; + if (value < min) min = value; + if (value > max) max = value; + } + // A pool bucket missing a device is a gap, as in `sumPowerAt`, not a dip. + if (count === 0 || (pool && count < rows.length)) continue; + const timeMs = bucketTimeMs(series, column); + samples.push({ + trace, + color, + overlayIndex, + column, + timeMs, + x: xMode === 'wall' ? timeMs : series.t[column] - series.t[0], + y: pool ? sum : sum / count, + min, + max, + gpuCount: count, + phase: windowPhase(trace, timeMs), + pool, + }); + } + return lttbDownsample( + samples, + HIT_POINTS_PER_TRACE, + (sample) => sample.x, + (sample) => sample.y, + ); +} + +type AnyContinuousScale = d3.ScaleLinear; + +function drawTraces( + group: d3.Selection, + xScale: AnyContinuousScale, + yScale: AnyContinuousScale, + model: DrawModel, + highlight: string | null, +): void { + const line = d3 + .line() + .defined((point) => point.y !== null) + .x((point) => xScale(point.x)) + .y((point) => yScale(point.y ?? 0)) + .curve(d3.curveLinear); + const selection = group + .selectAll('path.power-trace') + .data(model.paths, (path) => path.id); + selection.exit().remove(); + selection + .enter() + .append('path') + .attr('class', 'power-trace') + .attr('fill', 'none') + .attr('stroke-linejoin', 'round') + .attr('stroke-linecap', 'round') + .merge(selection) + .attr('data-trace-key', (path) => path.traceKey) + .attr('data-hw', (path) => path.hwKey) + .attr('data-segment', (path) => path.segment) + .attr('data-run-index', (path) => (path.overlayIndex === null ? null : path.overlayIndex)) + .attr('data-pool', (path) => path.pool ?? null) + .attr('stroke-dasharray', (path) => (path.pool ? poolDash(path.pool) : null)) + .attr('stroke', (path) => path.color) + .attr('stroke-width', (path) => path.width) + .attr('opacity', (path) => traceOpacity(path, highlight)) + .attr('d', (path) => line(path.points)); +} + +function traceOpacity(path: TracePath, highlight: string | null): number { + if (highlight === null) return path.opacity; + return highlight === path.hwKey || highlight === path.traceKey + ? Math.min(1, path.opacity + 0.15) + : path.opacity * 0.15; +} + +function drawLabels( + group: d3.Selection, + xScale: AnyContinuousScale, + yScale: AnyContinuousScale, + model: DrawModel, + plotWidth: number, + highlight: string | null, +): void { + const selection = group + .selectAll('text.power-trace-label') + .data(model.labels, (label) => label.id); + selection.exit().remove(); + selection + .enter() + .append('text') + .attr('class', 'power-trace-label') + .attr('font-family', CHART_FONT_SANS) + .attr('font-size', px(CHART_TYPE.dataLabel)) + .attr('font-weight', '600') + .attr('dominant-baseline', 'middle') + .attr('pointer-events', 'none') + .merge(selection) + .attr('data-hw', (label) => label.hwKey) + .attr('data-pool', (label) => label.pool ?? null) + .attr('fill', (label) => label.color) + .attr('opacity', (label) => + highlight === null || highlight === label.hwKey || highlight === label.traceKey ? 1 : 0.2, + ) + .text((label) => label.text) + .each(function (label) { + const x = xScale(label.x); + const y = yScale(label.y); + // Sit just past the last sample; flip inside the plot near the right edge. + const overflow = x + 6 + label.text.length * 6.5 > plotWidth; + d3.select(this) + .attr('text-anchor', overflow ? 'end' : 'start') + .attr('x', overflow ? x - 6 : x + 6) + .attr('y', y); + }); +} + +function drawReferenceLines( + group: d3.Selection, + yScale: AnyContinuousScale, + width: number, + lines: ReferenceLine[], +): void { + group.selectAll('.power-reference').remove(); + const slots = referenceLabelSlots(lines); + lines.forEach((line, index) => { + const y = yScale(line.watts); + const g = group + .append('g') + .attr('class', 'power-reference') + .attr('data-reference', line.kind) + .attr('data-pool', line.pools?.join(' ') ?? null) + .attr('data-watts', line.watts); + g.append('line') + .attr('x1', 0) + .attr('x2', width) + .attr('y1', y) + .attr('y2', y) + .attr('stroke', line.color) + .attr('stroke-width', 1.25) + .attr('stroke-dasharray', line.kind === 'tdp' ? '6,4' : '2,4') + .attr('opacity', 0.9); + g.append('text') + .attr('x', width - 4) + .attr('y', y - 5 - slots[index] * REFERENCE_LABEL_ROW) + .attr('text-anchor', 'end') + .attr('fill', line.color) + .attr('font-family', CHART_FONT_SANS) + .attr('font-size', px(CHART_TYPE.annotation)) + .attr('font-weight', '600') + .text(line.label); + }); +} + +/** Reasons, then the undrawn configs (named when few, counted per hardware when many). */ +function describeMissing( + missing: readonly MissingTrace[], + t: (typeof STRINGS)[keyof typeof STRINGS], + hardwareLabel: (point: InferenceData) => string, +): string { + const reasons = new Map(); + for (const { reason } of missing) reasons.set(reason, (reasons.get(reason) ?? 0) + 1); + const reasonText = [...reasons.entries()] + .map(([reason, count]) => t.missingReason[reason](count)) + .join('; '); + let list: string; + if (missing.length <= MAX_LISTED_MISSING) { + list = missing + .map(({ point }) => `${hardwareLabel(point)} ${traceConfigLabel(point)}`) + .join(' · '); + } else { + const perHardware = new Map(); + for (const { point } of missing) { + const label = hardwareLabel(point); + perHardware.set(label, (perHardware.get(label) ?? 0) + 1); + } + list = [...perHardware.entries()].map(([label, count]) => `${label} ×${count}`).join(' · '); + } + return `${reasonText}. ${t.missingUndrawn}: ${list}`; +} + +export default function PowerTimeline({ + chartId, + data, + overlayData, + yLabel, + caption, +}: PowerTimelineProps) { + const locale = useLocale(); + const t = STRINGS[locale]; + const { hardwareConfig, hwTypesWithData } = useInferenceData(); + const { activeHwTypes, selectedPrecisions, quickFilters } = useInferenceFilters(); + const { isLegendExpanded, highContrast } = useInferenceDisplay(); + const { setBestPerSku, toggleHwType, setIsLegendExpanded } = useInferenceActions(); + const { + unofficialRunInfos, + runIndexByUrl, + activeOverlayHwTypes, + localOfficialOverride, + setUnifiedOverlaySelection, + } = useUnofficialRun(); + + const [xModeChoice, setXModeChoice] = useState(null); + const [lineMode, setLineMode] = useState('mean'); + const [showUtility, setShowUtility] = useState(false); + const [highlight, setHighlight] = useState(null); + /** Trace a "View power trace" deep link asked for, once the join has produced it. */ + const [focusKey, setFocusKey] = useState(null); + /** The deep-link request, read once on mount; `undefined` until read, `null` once honoured. */ + const requestedFocusRef = useRef(undefined); + + // The chart's point list still carries every precision, quick-filtered rows + // and rows without a validated average (ScatterGraph applies those gates at + // draw time); only rows that plot on the measured-average axis for the + // current selection have a trace to look up. + const plotsHere = useCallback( + (point: InferenceData) => + point.measuredPowerTimeline !== undefined && + selectedPrecisions.includes(point.precision) && + matchesQuickFilters(point, quickFilters), + [selectedPrecisions, quickFilters], + ); + const measuredData = useMemo(() => data.filter(plotsHere), [data, plotsHere]); + const overlayPoints = useMemo( + () => (overlayData?.data ?? []).filter(plotsHere), + [overlayData, plotsHere], + ); + const overlayPointSet = useMemo(() => new Set(overlayPoints), [overlayPoints]); + const allPoints = useMemo( + () => [...measuredData, ...overlayPoints], + [measuredData, overlayPoints], + ); + + const hwKeysInData = useMemo( + () => + [...new Set(measuredData.map((point) => point.hwKey))].toSorted( + (a, b) => getModelSortIndex(a) - getModelSortIndex(b) || a.localeCompare(b), + ), + [measuredData], + ); + const stableHcKeys = useMemo(() => [...hwTypesWithData], [hwTypesWithData]); + const activeOfficialKeys = useMemo(() => [...activeHwTypes], [activeHwTypes]); + const { resolveColor, getCssColor } = useThemeColors({ + highContrast, + identifiers: hwKeysInData, + activeKeys: activeOfficialKeys, + hcKeys: stableHcKeys, + }); + + // ── Telemetry fetch: one request per workflow run ────────────────────────── + const requests = useMemo(() => planPowerTimelineRequests(allPoints), [allPoints]); + // The deep-link request is read once, before planning, so its run is fetched + // even when the chart spans more runs than the cap. + if (requestedFocusRef.current === undefined) { + requestedFocusRef.current = consumePowerTraceFocus(); + } + const focusRunRef = useRef(traceKeyRunId(requestedFocusRef.current)); + // Overlay runs were requested explicitly (`?unofficialrun=`), so they take + // the cap's slots before official runs; the deep-linked run still goes first. + const overlayRunIds = useMemo( + () => + new Set( + overlayPoints + .map((point) => runIdFromUrl(point.run_url)) + .filter((runId): runId is string => runId !== null), + ), + [overlayPoints], + ); + const fetchedRequests = useMemo( + () => + prioritizeRun(prioritizeRuns(requests, overlayRunIds), focusRunRef.current).slice( + 0, + POWER_TIMELINE_MAX_RUNS, + ), + [requests, overlayRunIds], + ); + const droppedRuns = requests.length - fetchedRequests.length; + const queries = useQueries({ + queries: fetchedRequests.map((request) => ({ + queryKey: ['power-timeline', request.runId, request.prefix, request.sources] as const, + queryFn: ({ signal }: { signal: AbortSignal }) => fetchPowerSeries(request, signal), + staleTime: 0, + refetchOnWindowFocus: true, + retry: 1, + })), + }); + const loadingRuns = queries.filter((query) => query.isPending).length; + const errors = fetchedRequests + .map((request, index) => ({ request, error: queries[index].error })) + .filter( + (entry): entry is { request: PowerTimelineRequest; error: Error } => + entry.error instanceof Error, + ); + const responses = useMemo(() => { + const map = new Map(); + fetchedRequests.forEach((request, index) => { + const response = queries[index].data; + if (response) map.set(request.runId, response); + }); + return map; + // Query data objects are stable per fetch; deriving from them keeps the map memoised. + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [fetchedRequests, ...queries.map((query) => query.data)]); + + const { traces, missing } = useMemo( + () => joinPowerTimeline(allPoints, responses), + [allPoints, responses], + ); + const hasAnyArtifact = requests.length > 0; + + // Honour the deep link once its trace exists; a disaggregated trace opens in + // pool mode because its prefill / decode split is what the reader came for. + useEffect(() => { + const requested = requestedFocusRef.current; + if (!requested) return; + const trace = traces.find((entry) => entry.key === requested); + if (!trace) return; + requestedFocusRef.current = null; + setFocusKey(trace.key); + if (tracePools(trace.series).length > 0) setLineMode('pool'); + }, [traces]); + + useEffect(() => { + if (loadingRuns > 0 || responses.size === 0) return; + track('inference_power_timeline_loaded', { + traces: traces.length, + missing: missing.length, + runs: responses.size, + }); + }, [loadingRuns, responses.size, traces.length, missing.length]); + + // ── Visible traces and their colours ──────────────────────────────────────── + const colorForTrace = useCallback( + (trace: PowerTimelineTrace): { color: string; overlayIndex: number | null } => { + if (overlayPointSet.has(trace.point)) { + const index = overlayRunIndex(trace.point.run_url ?? null, runIndexByUrl); + return { color: overlayRunColor(index), overlayIndex: index }; + } + return { color: getCssColor(resolveColor(trace.point.hwKey)), overlayIndex: null }; + }, + [overlayPointSet, runIndexByUrl, getCssColor, resolveColor], + ); + // Same visibility source as ScatterGraph: an overlay session may hold a + // local official selection that has not been written back to the filters. + const officialHwTypes = localOfficialOverride ?? activeHwTypes; + // With an overlay loaded the chart reads localOfficialOverride, so a legend + // click must write the unified selection the way ScatterGraph does; the + // context's toggleHwType would change activeHwTypes with no visible effect. + const handleToggleHwType = useCallback( + (key: string) => { + if (!overlayData) { + toggleHwType(key); + return; + } + setBestPerSku(false, { applySelection: false }); + const official = new Set([...officialHwTypes].filter((hw) => hwTypesWithData.has(hw))); + setUnifiedOverlaySelection( + computeToggle(official, key, hwTypesWithData), + activeOverlayHwTypes, + ); + }, + [ + overlayData, + toggleHwType, + setBestPerSku, + officialHwTypes, + hwTypesWithData, + setUnifiedOverlaySelection, + activeOverlayHwTypes, + ], + ); + const visibleTraces = useMemo( + () => + traces.filter((trace) => + overlayPointSet.has(trace.point) + ? activeOverlayHwTypes.has(trace.point.hwKey) + : officialHwTypes.has(trace.point.hwKey), + ), + [traces, overlayPointSet, activeOverlayHwTypes, officialHwTypes], + ); + // Focus follows visibility: hiding the focused hardware in the legend lifts + // the dimming and the chip instead of dimming everything with nothing lit. + const focusedTrace = useMemo( + () => visibleTraces.find((trace) => trace.key === focusKey) ?? null, + [visibleTraces, focusKey], + ); + /** Legend hover wins over the deep-link focus while it lasts. */ + const activeHighlight = highlight ?? focusedTrace?.key ?? null; + const visibleRunCount = useMemo( + () => new Set(visibleTraces.map((trace) => trace.runId)).size, + [visibleTraces], + ); + const xMode: XMode = xModeChoice ?? (visibleRunCount <= 1 ? 'wall' : 'elapsed'); + + // The pools switch is offered only where a visible trace carries worker roles; + // pool mode left without one would draw deployment totals with no way back. + const hasPools = useMemo( + () => visibleTraces.some((trace) => tracePools(trace.series).length > 0), + [visibleTraces], + ); + useEffect(() => { + if (lineMode === 'pool' && !hasPools && visibleTraces.length > 0) setLineMode('mean'); + }, [lineMode, hasPools, visibleTraces.length]); + + const model = useMemo(() => { + const paths: TracePath[] = []; + const labels: TraceLabel[] = []; + for (const trace of visibleTraces) { + const { color, overlayIndex } = colorForTrace(trace); + const tracePathSet = tracePaths(trace, color, overlayIndex, xMode, lineMode); + paths.push(...tracePathSet); + if (visibleTraces.length > MAX_LABELED_TRACES) continue; + // One end label per trace; per pool in pool mode, so the role reads off the line. + const groups: { pool?: PowerPoolRole; paths: TracePath[] }[] = + lineMode === 'pool' + ? drawnPools(trace.series).map((pool) => ({ + pool: pool.role, + paths: tracePathSet.filter((path) => path.pool === pool.role), + })) + : [{ paths: tracePathSet }]; + for (const group of groups) { + const anchor = + group.paths.find((path) => path.segment === 'window') ?? + group.paths.find((path) => path.segment === 'full'); + const last = anchor?.points.filter((point) => point.y !== null).at(-1); + if (!last || last.y === null) continue; + labels.push({ + id: group.pool ? `${trace.key}:${group.pool}` : trace.key, + traceKey: trace.key, + hwKey: trace.point.hwKey, + pool: group.pool, + color, + text: group.pool + ? `c${trace.point.conc} · ${t.poolShort[group.pool]}` + : `c${trace.point.conc}`, + x: last.x, + y: last.y, + }); + } + } + return { paths, labels }; + }, [visibleTraces, colorForTrace, xMode, lineMode, t]); + + const samples = useMemo( + () => + visibleTraces.flatMap((trace) => { + const { color, overlayIndex } = colorForTrace(trace); + return traceSamples(trace, color, overlayIndex, xMode, lineMode); + }), + [visibleTraces, colorForTrace, xMode, lineMode], + ); + + // Rated references: per hardware in mean / per-GPU modes; per (hardware, + // pool size) in pool mode, scaled to the pool so the summed line and its + // ceiling share the axis. Roles of one hardware that hold the same number + // of GPUs share a ceiling and draw as one line (`prefill / decode ×16`). + const referenceLines = useMemo(() => { + const lines: ReferenceLine[] = []; + const pushLines = (base: string, color: string, pool?: PoolSizeGroup) => { + const specs = HW_REGISTRY[base]; + if (!specs) return; + const label = specs.label ?? base.toUpperCase(); + const size = pool?.size ?? 1; + const id = pool ? `${base}:${pool.roles.join('+')}:${pool.size}` : base; + const name = pool + ? `${label} ${pool.roles.map((role) => t.poolShort[role]).join(' / ')} ×${pool.size}` + : label; + if (specs.tdp > 0) { + lines.push({ + id: `tdp:${id}`, + watts: specs.tdp * size, + label: `${name} ${t.tdp} ${specs.tdp * size} W`, + color, + kind: 'tdp', + pools: pool?.roles, + }); + } + if (showUtility && specs.power > 0) { + const watts = pool ? Math.round(specs.power * 1000) * size : specs.power * 1000; + lines.push({ + id: `utility:${id}`, + watts, + label: `${name} ${t.allIn} ${Math.round(watts)} W`, + color, + kind: 'utility', + pools: pool?.roles, + }); + } + }; + // First trace of a hardware sets the reference colour, as before. + const perBase = new Map(); + for (const trace of visibleTraces) { + const base = baseHardware(trace.point.hwKey); + if (!perBase.has(base)) perBase.set(base, { color: colorForTrace(trace).color, pools: [] }); + if (lineMode === 'pool') perBase.get(base)!.pools.push(...drawnPools(trace.series)); + } + for (const [base, { color, pools }] of perBase) { + if (lineMode !== 'pool') { + pushLines(base, color); + continue; + } + for (const group of groupPoolsBySize(pools)) pushLines(base, color, group); + } + return lines; + }, [visibleTraces, colorForTrace, showUtility, lineMode, t]); + + // ── Scales ───────────────────────────────────────────────────────────────── + const xDomain = useMemo<[number, number]>(() => { + let min = Number.POSITIVE_INFINITY; + let max = Number.NEGATIVE_INFINITY; + for (const path of model.paths) { + if (path.segment !== 'full') continue; + for (const point of path.points) { + if (point.x < min) min = point.x; + if (point.x > max) max = point.x; + } + } + if (!Number.isFinite(min) || !Number.isFinite(max)) { + return xMode === 'wall' ? [Date.UTC(2026, 0, 1), Date.UTC(2026, 0, 1, 0, 10)] : [0, 600]; + } + return min === max ? [min, max + (xMode === 'wall' ? 60_000 : 60)] : [min, max]; + }, [model.paths, xMode]); + const yDomain = useMemo<[number, number]>(() => { + let max = 0; + for (const path of model.paths) { + for (const point of path.points) if (point.y !== null && point.y > max) max = point.y; + } + for (const line of referenceLines) if (line.watts > max) max = line.watts; + return [0, max > 0 ? max * 1.06 : 100]; + }, [model.paths, referenceLines]); + + const xTickFormat = useMemo(() => { + if (xMode === 'elapsed') return (value: d3.AxisDomain) => formatElapsed(Number(value)); + const span = xDomain[1] - xDomain[0]; + const crossesDate = formatUtcDate(new Date(xDomain[0])) !== formatUtcDate(new Date(xDomain[1])); + const format = d3.utcFormat( + crossesDate ? '%m/%d %H:%M' : span < 3 * 60_000 ? '%H:%M:%S' : '%H:%M', + ); + return (value: d3.AxisDomain) => + format(value instanceof Date ? value : new Date(Number(value))); + }, [xMode, xDomain]); + + // ── Layers ───────────────────────────────────────────────────────────────── + const highlightRef = useRef(activeHighlight); + highlightRef.current = activeHighlight; + const layers = useMemo[]>( + () => [ + { + type: 'custom', + key: 'power-reference-lines', + render: (group, ctx) => { + drawReferenceLines(group, ctx.yScale as AnyContinuousScale, ctx.width, referenceLines); + }, + }, + { + type: 'custom', + key: 'power-traces', + render: (group, ctx: RenderContext) => { + drawTraces( + group, + ctx.xScale as AnyContinuousScale, + ctx.yScale as AnyContinuousScale, + model, + highlightRef.current, + ); + }, + onZoom: (group, ctx: ZoomContext) => { + drawTraces( + group, + ctx.newXScale as AnyContinuousScale, + ctx.newYScale as AnyContinuousScale, + model, + highlightRef.current, + ); + }, + }, + { + type: 'point', + key: 'power-hit-points', + data: samples, + config: { + getCx: () => 0, + getCy: () => 0, + getX: (sample) => sample.x, + getY: (sample) => sample.y, + getColor: (sample) => sample.color, + getRadius: () => 2, + // Pool streams of one trace share columns; the role keeps their keys apart. + keyFn: (sample) => + sample.pool + ? `${sample.trace.key}:${sample.pool.role}:${sample.column}` + : `${sample.trace.key}:${sample.column}`, + maxPoints: Number.POSITIVE_INFINITY, + }, + }, + { + type: 'custom', + key: 'power-trace-labels', + render: (group, ctx: RenderContext) => { + drawLabels( + group, + ctx.xScale as AnyContinuousScale, + ctx.yScale as AnyContinuousScale, + model, + ctx.width, + highlightRef.current, + ); + }, + onZoom: (group, ctx: ZoomContext) => { + drawLabels( + group, + ctx.newXScale as AnyContinuousScale, + ctx.newYScale as AnyContinuousScale, + model, + ctx.width, + highlightRef.current, + ); + }, + }, + ], + [referenceLines, model, samples], + ); + + const onDisplayUpdate = useCallback( + (ctx: RenderContext) => { + const root = d3.select(ctx.layout.svg.node() as SVGSVGElement); + root + .selectAll('path.power-trace') + .attr('opacity', (path) => traceOpacity(path, activeHighlight)); + root + .selectAll('text.power-trace-label') + .attr('opacity', (label) => + activeHighlight === null || + activeHighlight === label.hwKey || + activeHighlight === label.traceKey + ? 1 + : 0.2, + ); + }, + [activeHighlight], + ); + + const hardwareLabel = useCallback( + (point: InferenceData): string => { + const config = overlayPointSet.has(point) + ? overlayData?.hardwareConfig[point.hwKey] + : hardwareConfig[point.hwKey]; + return config ? getDisplayLabel(config) : point.hwKey; + }, + [overlayPointSet, overlayData, hardwareConfig], + ); + + const tooltipContent = useCallback( + (sample: TimelineSample, isPinned: boolean) => { + const { trace } = sample; + const point = trace.point; + const overlayInfo = + sample.overlayIndex === null ? null : unofficialRunInfos[sample.overlayIndex]; + const tdp = HW_REGISTRY[baseHardware(point.hwKey)]?.tdp ?? 0; + const elapsed = formatElapsed((sample.timeMs - trace.series.startMs) / 1000); + const clock = `${formatUtcClock(new Date(sample.timeMs))} UTC`; + const time = + xMode === 'wall' ? `${clock} · +${elapsed} ${t.sinceStart}` : `+${elapsed} · ${clock}`; + const colon = locale === 'zh' ? ':' : ':'; + const validated = point.measuredAvgPower?.y; + const { pool } = sample; + const readings = pool + ? `
    ${t.pool}${colon} ${t.poolShort[pool.role]} · ${t.gpus(pool.gpuCount)}
    +
    ${t.poolPower}${colon} ${formatWatts(sample.y)} W${ + tdp > 0 + ? ` (${((sample.y / (tdp * pool.gpuCount)) * 100).toFixed(0)}% ${t.poolTdp})` + : '' + }
    +
    ${t.meanPerGpu}${colon} ${(sample.y / sample.gpuCount).toFixed(1)} W · ${t.min} ${sample.min.toFixed(1)} W · ${t.max} ${sample.max.toFixed(1)} W
    ` + : `
    ${t.meanPerGpu}${colon} ${sample.y.toFixed(1)} W${ + tdp > 0 + ? ` (${((sample.y / tdp) * 100).toFixed(0)}% ${t.tdp})` + : '' + }
    +
    ${t.gpus(sample.gpuCount)} · ${t.min} ${sample.min.toFixed(1)} W · ${t.max} ${sample.max.toFixed(1)} W
    `; + return `
    + ${isPinned ? `
    ${t.dismiss}
    ` : ''} +
    ${hardwareLabel(point)} · ${traceConfigLabel(point)}${ + overlayInfo ? ` · ✕ ${overlayInfo.branch || `run ${overlayInfo.id}`}` : '' + }
    +
    ${time}
    + ${readings} +
    ${t.phase[sample.phase]}
    + ${ + typeof validated === 'number' + ? `
    ${t.validated}${colon} ${validated.toFixed(1)} W
    ` + : '' + } +
    `; + }, + [unofficialRunInfos, xMode, t, locale, hardwareLabel], + ); + + // ── Legend ───────────────────────────────────────────────────────────────── + const legendItems = useMemo(() => { + const overlayItems = + overlayData && unofficialRunInfos.length > 0 + ? unofficialRunInfos + .map((info, index) => { + const hasPoints = overlayPoints.some( + (point) => overlayRunIndex(point.run_url ?? null, runIndexByUrl) === index, + ); + if (!hasPoints) return null; + const branch = info.branch || `run ${info.id}`; + return { + name: `✕ unofficial-run-${info.id}`, + label: `✕ ${branch}`, + color: overlayRunColor(index), + title: `${t.unofficialRun}: ${branch}`, + isHighlighted: true, + hw: `overlay-run-${info.id}`, + isActive: true, + isRemovable: false, + onClick: () => {}, + tooltip: ( +
    +
    {t.unofficialRun}
    +
    + {t.branch}: {branch} +
    + {info.url && ( + + {t.viewWorkflow} + + )} +
    + ), + }; + }) + .filter((item): item is NonNullable => item !== null) + : []; + const officialItems = hwKeysInData + .filter((key) => hwTypesWithData.has(key) && hardwareConfig[key]) + .map((key) => { + const config = hardwareConfig[key]; + return { + name: config.name, + label: getDisplayLabel(config), + color: resolveColor(key), + title: config.gpu, + hw: key, + isActive: officialHwTypes.has(key), + onClick: () => { + handleToggleHwType(key); + track('latency_hw_type_toggled', { hw: key }); + }, + tooltip: null, + }; + }); + return [...overlayItems, ...officialItems]; + }, [ + overlayData, + unofficialRunInfos, + overlayPoints, + runIndexByUrl, + hwKeysInData, + hwTypesWithData, + hardwareConfig, + resolveColor, + officialHwTypes, + handleToggleHwType, + t, + ]); + + // Per-GPU and pools are two views of the same lines, so either switch turns + // the other off; both fall back to the mean. + const chooseLineMode = (next: LineMode) => { + setLineMode(next); + track('inference_power_timeline_lines_changed', { lines: next }); + }; + const switches: LegendSwitchConfig[] = [ + { + id: 'power-timeline-per-gpu', + label: t.perGpu, + checked: lineMode === 'gpu', + onCheckedChange: (checked) => chooseLineMode(checked ? 'gpu' : 'mean'), + infoTooltip: t.perGpuHelp, + }, + ]; + if (hasPools) { + switches.push({ + id: 'power-timeline-pools', + label: t.pools, + checked: lineMode === 'pool', + onCheckedChange: (checked) => chooseLineMode(checked ? 'pool' : 'mean'), + infoTooltip: t.poolsHelp, + }); + } + switches.push({ + id: 'power-timeline-utility', + label: t.utilityLines, + checked: showUtility, + onCheckedChange: (checked) => { + setShowUtility(checked); + track('inference_power_timeline_utility_toggled', { enabled: checked }); + }, + infoTooltip: t.utilityHelp, + }); + + const legendElement = ( + { + setIsLegendExpanded(expanded); + track('latency_legend_expanded', { expanded }); + }} + onItemHover={(id) => setHighlight(id)} + onItemHoverEnd={() => setHighlight(null)} + hideAtomFootnote + switches={switches} + /> + ); + + const runInfos = useMemo( + () => + fetchedRequests + .map((request) => responses.get(request.runId)?.runInfo) + .filter((info): info is NonNullable => Boolean(info)), + [fetchedRequests, responses], + ); + + const toolbar = ( +
    + {t.timeAxis} + + value={xMode} + ariaLabel={t.timeAxis} + role="group" + options={[ + { value: 'wall', label: t.wall, testId: 'power-timeline-axis-wall' }, + { value: 'elapsed', label: t.elapsed, testId: 'power-timeline-axis-elapsed' }, + ]} + onValueChange={(mode) => { + setXModeChoice(mode); + track('inference_power_timeline_axis_changed', { mode }); + }} + /> +
    + ); + + let emptyMessage: string | null = null; + if (loadingRuns > 0) emptyMessage = t.loading(loadingRuns); + else if (!hasAnyArtifact) emptyMessage = t.noArtifacts; + else if (visibleTraces.length === 0) emptyMessage = t.noTraces; + + return ( +
    + + key={`${chartId}-${xMode}-${lineMode}`} + chartId={chartId} + data={samples} + height={CHART_HEIGHT} + margin={MARGIN} + watermark={overlayData && overlayPoints.length > 0 ? 'unofficial' : 'logo'} + testId="power-timeline-chart-svg" + grabCursor + instructions={t.instructions} + xScale={ + xMode === 'wall' + ? { type: 'time', domain: [new Date(xDomain[0]), new Date(xDomain[1])] } + : { type: 'linear', domain: xDomain } + } + yScale={{ type: 'linear', domain: yDomain, nice: true }} + xAxis={{ + label: xMode === 'wall' ? t.xWall : t.xElapsed, + tickValues: (scale) => { + const timeScale = scale as + | d3.ScaleTime + | d3.ScaleLinear; + const [left, right] = timeScale.range(); + return timeScale.ticks(Math.max(2, Math.min(10, Math.floor((right - left) / 80)))); + }, + tickFormat: xTickFormat, + }} + yAxis={{ label: lineMode === 'pool' ? t.yPool : yLabel, tickCount: 8 }} + layers={layers} + displayIdentity={activeHighlight ?? ''} + onDisplayUpdate={onDisplayUpdate} + zoom={{ + enabled: true, + axes: 'x', + scaleExtent: [1, 60], + resetEventName: `power_timeline_zoom_reset_${chartId}`, + }} + tooltip={{ + rulerType: 'crosshair', + content: tooltipContent, + getRulerX: (sample, xScale) => (xScale as AnyContinuousScale)(sample.x), + getRulerY: (sample, yScale) => yScale(sample.y), + onHoverStart: (selection) => { + selection.attr('r', 5).attr('stroke', 'white').attr('stroke-width', 1); + }, + onHoverEnd: (selection) => { + selection.attr('r', 2).attr('stroke', 'none'); + }, + attachToLayer: 2, + }} + legendElement={legendElement} + caption={ + <> + {caption} + {toolbar} + + } + noDataOverlay={ + emptyMessage ? ( +
    + {emptyMessage} +
    + ) : undefined + } + /> +
    + {focusedTrace && ( +

    + + {t.focused( + `${hardwareLabel(focusedTrace.point)} · ${traceConfigLabel(focusedTrace.point)}`, + )} + + +

    + )} + {errors.map(({ request, error }) => ( +

    + {t.loadError(request.runId, error.message)} +

    + ))} + {droppedRuns > 0 &&

    {t.droppedRuns(droppedRuns)}

    } + {loadingRuns === 0 && missing.length > 0 && traces.length > 0 && ( +

    + {t.missing(missing.length, traces.length + missing.length)}{' '} + {describeMissing(missing, t, hardwareLabel)} +

    + )} + {runInfos.length > 0 && ( +

    + {t.telemetry}:{' '} + {runInfos.map((info, index) => ( + + {index > 0 && ' · '} + + {`run ${info.id}`} + + {info.createdAt ? ` (${formatUtcDate(new Date(info.createdAt))})` : ''} + + ))} +

    + )} +

    {t.method}

    + {hasPools &&

    {t.methodPools}

    } +
    +
    + ); +} diff --git a/packages/app/src/components/inference/utils/powerTimeline.test.ts b/packages/app/src/components/inference/utils/powerTimeline.test.ts new file mode 100644 index 000000000..409a74ad1 --- /dev/null +++ b/packages/app/src/components/inference/utils/powerTimeline.test.ts @@ -0,0 +1,21 @@ +import { describe, expect, it } from 'vitest'; + +import { prioritizeRuns } from './powerTimeline'; + +describe('prioritizeRuns', () => { + const requests = ['1', '2', '3', '4', '5'].map((runId) => ({ + runId, + prefix: '', + sources: [], + })); + + it('moves overlay runs ahead of official runs and keeps both orders', () => { + expect(prioritizeRuns(requests, new Set(['5', '3'])).map((request) => request.runId)).toEqual([ + '3', + '5', + '1', + '2', + '4', + ]); + }); +}); diff --git a/packages/app/src/components/inference/utils/powerTimeline.ts b/packages/app/src/components/inference/utils/powerTimeline.ts new file mode 100644 index 000000000..cff71ac26 --- /dev/null +++ b/packages/app/src/components/inference/utils/powerTimeline.ts @@ -0,0 +1,344 @@ +/** + * Joins chart points to the per-second GPU telemetry behind their measured + * average power (the PowerX "Timeline" display). + * + * Every validated row records its window in `power_audit`, whose `source` is + * `power_validation_.json`. Two collectors publish the telemetry: + * - single-node runners upload one `gpu_metrics_` CSV artifact per + * config, so the source names the artifact exactly; + * - Slurm / Dynamo runners upload one `power_audit_` bundle per + * sweep whose `LOGS/power/samples.csv` covers every concurrency; the API cuts + * it into one series per `power_validation_*.json` it contains and labels + * each with that `source`, so the same file name joins it to the row. + * Nothing is matched by hardware or concurrency; points whose telemetry is + * missing (expired artifact, another collector) are reported, not guessed. + */ +import type { + GpuPowerRole, + GpuPowerSeries, + GpuPowerSeriesResponse, +} from '@/components/gpu-power/power-series'; +import type { InferenceData } from '@/components/inference/types'; + +export const POWER_TIMELINE_METRIC_KEY = 'y_measuredPowerTimeline'; +const ARTIFACT_PREFIX = 'gpu_metrics_'; +const SOURCE_PATTERN = /^(?:.*\/)?power_validation_(?.+)\.json$/u; + +interface AuditedPoint { + power_audit?: { source?: string } | null; +} + +/** The `` of a point's `power_validation_.json` audit source. */ +export function telemetryNameForPoint(point: AuditedPoint): string | null { + const source = point.power_audit?.source; + if (!source) return null; + return SOURCE_PATTERN.exec(source)?.groups?.name ?? null; +} + +/** `gpu_metrics_` for a point, from its power-audit source. */ +export function telemetryArtifactForPoint(point: AuditedPoint): string | null { + const name = telemetryNameForPoint(point); + return name ? `${ARTIFACT_PREFIX}${name}` : null; +} + +/** + * Stable identity of a point's trace: run id plus audit name. Unique within a + * chart because the audit name carries the config and the concurrency. + */ +export function traceKeyForPoint(point: AuditedPoint & { run_url?: string }): string | null { + const runId = runIdFromUrl(point.run_url); + const name = telemetryNameForPoint(point); + return runId && name ? `${runId}:${name}` : null; +} + +/** Workflow run id from a GitHub Actions run URL. */ +export function runIdFromUrl(url: string | null | undefined): string | null { + return url?.match(/\/runs\/(?\d+)/u)?.groups?.runId ?? null; +} + +export function longestCommonPrefix(values: readonly string[]): string { + if (values.length === 0) return ''; + let prefix = values[0]; + for (const value of values) { + let end = 0; + while (end < prefix.length && end < value.length && prefix[end] === value[end]) end++; + prefix = prefix.slice(0, end); + if (prefix === '') break; + } + return prefix; +} + +export interface PowerTimelineRequest { + runId: string; + /** RESULT_FILENAME prefix shared by every wanted artifact of the run. */ + prefix: string; + /** Sorted, unique validation basenames expected by the displayed points. */ + sources: string[]; +} + +/** + * One request per workflow run, narrowed to the common RESULT_FILENAME prefix + * of the points' artifacts so a sweep of other models is not downloaded. + * Points without a run URL or an audit source plan nothing. + */ +export function planPowerTimelineRequests( + points: readonly InferenceData[], +): PowerTimelineRequest[] { + const byRun = new Map>(); + for (const point of points) { + const runId = runIdFromUrl(point.run_url); + const artifact = telemetryArtifactForPoint(point); + if (!runId || !artifact) continue; + if (!byRun.has(runId)) byRun.set(runId, new Set()); + byRun.get(runId)!.add(artifact); + } + return [...byRun.entries()] + .toSorted(([a], [b]) => a.localeCompare(b)) + .map(([runId, artifacts]) => { + const names = [...artifacts].toSorted(); + return { + runId, + prefix: longestCommonPrefix(names.map((name) => name.slice(ARTIFACT_PREFIX.length))), + sources: names.map((name) => `power_validation_${name.slice(ARTIFACT_PREFIX.length)}.json`), + }; + }); +} + +/** Run id half of a trace key (`traceKeyForPoint`). */ +export function traceKeyRunId(key: string | null | undefined): string | null { + const runId = key?.split(':')[0]; + return runId || null; +} + +/** + * Moves the request for `runId` to the front so it survives the per-chart run + * cap; the order of the other requests is kept. Returns the same array when + * nothing needs moving. + */ +export function prioritizeRun( + requests: PowerTimelineRequest[], + runId: string | null, +): PowerTimelineRequest[] { + const index = runId ? requests.findIndex((request) => request.runId === runId) : -1; + if (index <= 0) return requests; + return [requests[index], ...requests.slice(0, index), ...requests.slice(index + 1)]; +} + +/** + * Moves every request whose run is in `runIds` ahead of the others, keeping + * the relative order inside both groups. Unofficial-run overlays are loaded + * on purpose, so their telemetry must survive the per-chart run cap before + * official rows compete for the remaining slots. Returns the same array when + * nothing needs moving. + */ +export function prioritizeRuns( + requests: PowerTimelineRequest[], + runIds: ReadonlySet, +): PowerTimelineRequest[] { + if (runIds.size === 0) return requests; + const first = requests.filter((request) => runIds.has(request.runId)); + if (first.length === 0 || first.length === requests.length) return requests; + const rest = requests.filter((request) => !runIds.has(request.runId)); + const moved = first.some((request, index) => requests[index] !== request); + return moved ? [...first, ...rest] : requests; +} + +export interface PowerTimelineTrace { + /** `traceKeyForPoint(point)`. */ + key: string; + point: InferenceData; + runId: string; + series: GpuPowerSeries; + /** Validated measurement window (UTC ms) from `power_audit`, when recorded. */ + windowStartMs: number | null; + windowEndMs: number | null; +} + +/** + * Why a validated point has no trace: + * - `no-source`: the row predates `power_audit.source` (older schema), so no + * artifact can be named; + * - `no-run`: no workflow run URL to look in; + * - `run-not-fetched`: its run is not among the loaded responses (over the + * per-chart run limit, still loading, or the request failed); + * - `not-in-run`: the run was loaded but holds neither a matching + * `gpu_metrics_*` artifact nor a power-audit bundle with the point's + * validation file (expired, over the download cap, or another collector). + */ +export type MissingTraceReason = 'no-source' | 'no-run' | 'run-not-fetched' | 'not-in-run'; + +export interface MissingTrace { + point: InferenceData; + reason: MissingTraceReason; +} + +export interface PowerTimelineJoin { + traces: PowerTimelineTrace[]; + /** Points with a validated average but no telemetry trace, with the reason. */ + missing: MissingTrace[]; +} + +/** Attaches fetched series to points; order follows `points`. */ +export function joinPowerTimeline( + points: readonly InferenceData[], + responses: ReadonlyMap, +): PowerTimelineJoin { + const traces: PowerTimelineTrace[] = []; + const missing: MissingTrace[] = []; + for (const point of points) { + const runId = runIdFromUrl(point.run_url); + const name = telemetryNameForPoint(point); + if (!name) { + missing.push({ point, reason: 'no-source' }); + continue; + } + if (!runId) { + missing.push({ point, reason: 'no-run' }); + continue; + } + const response = responses.get(runId); + if (!response) { + missing.push({ point, reason: 'run-not-fetched' }); + continue; + } + // A bundle-cut series names the point's validation file; a per-config + // CSV artifact names the config itself. + const source = `power_validation_${name}.json`; + const artifact = `${ARTIFACT_PREFIX}${name}`; + const series = + response.series.find((entry) => entry.source === source) ?? + response.series.find((entry) => entry.source === undefined && entry.artifact === artifact); + if (!series) { + missing.push({ point, reason: 'not-in-run' }); + continue; + } + const audit = point.power_audit; + traces.push({ + key: `${runId}:${name}`, + point, + runId, + series, + windowStartMs: unixSecondsToMs(audit?.window_start_unix), + windowEndMs: unixSecondsToMs(audit?.window_end_unix), + }); + } + return { traces, missing }; +} + +function unixSecondsToMs(seconds: number | undefined): number | null { + return typeof seconds === 'number' && Number.isFinite(seconds) ? seconds * 1000 : null; +} + +/** Sample position relative to the validated window. */ +export type WindowPhase = 'before' | 'window' | 'after' | 'unknown'; + +export function windowPhase(trace: PowerTimelineTrace, timeMs: number): WindowPhase { + if (trace.windowStartMs === null || trace.windowEndMs === null) return 'unknown'; + if (timeMs < trace.windowStartMs) return 'before'; + if (timeMs > trace.windowEndMs) return 'after'; + return 'window'; +} + +/** Short config label: `TP8 · c64`, plus `PD` for disaggregated points. */ +export function traceConfigLabel(point: InferenceData): string { + const parts: string[] = []; + if (point.disagg) parts.push('PD'); + if (typeof point.tp === 'number' && point.tp > 0) parts.push(`TP${point.tp}`); + parts.push(`c${point.conc}`); + return parts.join(' · '); +} + +// ── Worker-role pools ──────────────────────────────────────────────────────── + +export type PowerPoolRole = GpuPowerRole | 'all'; + +export interface PowerPool { + role: PowerPoolRole; + /** Row indices into `series.power`, in series order. */ + rows: number[]; +} + +const POOL_ORDER: readonly PowerPoolRole[] = ['all', 'prefill', 'decode']; + +/** + * The GPU pools of a series by worker role, in `prefill`, `decode` order — + * only the roles that have at least one device. A series whose collector + * assigns no roles yields no pools; callers fall back to `allGpuPool`. + */ +export function tracePools(series: Pick): PowerPool[] { + const rows = new Map(); + series.devices?.forEach((device, row) => { + if (!device.role) return; + if (!rows.has(device.role)) rows.set(device.role, []); + rows.get(device.role)!.push(row); + }); + return POOL_ORDER.filter((role) => rows.has(role)).map((role) => ({ + role, + rows: rows.get(role)!, + })); +} + +export interface PoolSizeGroup { + size: number; + roles: PowerPoolRole[]; +} + +/** + * Pools that hold the same number of GPUs share one rated ceiling, so their + * reference draws once, labelled `prefill / decode ×16`, instead of two labels + * printed over each other. Sizes ascending, roles in pool order, deduplicated. + */ +export function groupPoolsBySize( + pools: readonly Pick[], +): PoolSizeGroup[] { + const bySize = new Map>(); + for (const pool of pools) { + if (!bySize.has(pool.rows.length)) bySize.set(pool.rows.length, new Set()); + bySize.get(pool.rows.length)!.add(pool.role); + } + return [...bySize.entries()] + .toSorted(([a], [b]) => a - b) + .map(([size, roles]) => ({ + size, + roles: POOL_ORDER.filter((role) => roles.has(role)), + })); +} + +/** + * Label row for each reference line: lines at the same watts (different + * hardware with an equal pool ceiling) stack their labels upward, slot 0 on + * the line and slot n `n` rows above, instead of overprinting. Input order. + */ +export function referenceLabelSlots(lines: readonly { watts: number }[]): number[] { + const used = new Map(); + return lines.map((line) => { + const slot = used.get(line.watts) ?? 0; + used.set(line.watts, slot + 1); + return slot; + }); +} + +/** Every GPU of the series as one pool. */ +export function allGpuPool(series: Pick): PowerPool { + return { role: 'all', rows: series.power.map((_, row) => row) }; +} + +// ── Deep link from a pinned scatter tooltip ───────────────────────────────── +// +// "View power trace" on a pinned tooltip switches the metric to the Timeline +// display; the timeline mounts afterwards and reads the requested trace here +// so it can emphasise that config. Module state rather than URL state: the +// focus is a one-shot gesture, and the share link stays `i_metric` alone. + +let pendingFocus: string | null = null; + +export function requestPowerTraceFocus(key: string): void { + pendingFocus = key; +} + +/** The pending focus request, cleared on read. */ +export function consumePowerTraceFocus(): string | null { + const key = pendingFocus; + pendingFocus = null; + return key; +} From 99841c9d075ce8fe007fceb494774a483ef8c2a3 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Tue, 29 Sep 2026 12:27:46 -0700 Subject: [PATCH 5/7] docs: PowerX permanent view, power boundaries and state ownership MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Document the gated power-boundary metrics and share-link state, the power-basis derived fields, uniform-host system power modeling, and the dashboard doc index entry. 中文:记录门控的功耗边界指标与分享链接状态、power-basis 派生字段、统一主机的系统功耗建模及 docs 索引条目。 --- docs/data-transforms.md | 21 ++- docs/powerx-permanent-view.md | 286 ++++++++++++++++++++++++++++++++++ docs/powerx-system-power.md | 23 +-- docs/state-ownership.md | 74 +++++++++ 4 files changed, 394 insertions(+), 10 deletions(-) create mode 100644 docs/powerx-permanent-view.md diff --git a/docs/data-transforms.md b/docs/data-transforms.md index 5fef22b39..f5dcd0935 100644 --- a/docs/data-transforms.md +++ b/docs/data-transforms.md @@ -64,7 +64,26 @@ Returns `{ chartData: InferenceData[][], hardwareConfig: HardwareConfig }`. - `tpPerMw` — `(tputPerGpu * 1000) / hardwarePower` (GPU power in kW, result in tok/s/MW). - Cost-per-million fields — GPU hourly cost divided by tokens-per-hour (in millions): `costh` / `costr` for owning-at-large-hyperscaler-volume / 3-year-rental pricing respectively. Three token variants exist: combined (`costh`/`costr`), output-only (`costhOutput`/`costrOutput`), and input-only (`costhi`/`costri`). The former Neocloud tier (`costn`/`costni`/`costnOutput`) was removed; its share-link keys alias to the hyperscaler-volume metric. - Infrastructure purchasing-power fields divide tokens per hour by the same hourly costs. Total (`tokensPerDollarH`/`R`), output-only (`outputTokensPerDollarH`/`R`), and input-only (`inputTokensPerDollarH`/`R`) are separate USD Y-axis metrics. The former ¥-priced `tokensPerRmb*` axes and the Neocloud `*PerDollarN` axes were removed; their share-link keys alias to the matching hyperscaler-volume $ metric. The dashboard defaults to the hyperscaler-volume total-tokens-per-dollar axis (`y_tokensPerDollarH`). Links created for the removed API-price `i_metric=y_tokensPerDollar` axis resolve to the same hyperscaler-volume variant. -- Energy fields — `jTotal` / `jOutput` / `jInput`: `(hardwarePower * 1000) / tputPerGpu` (Joules per token, where power in kW is converted to W). +- Energy fields — `jTotal` / `jOutput` / `jInput`: `(hardwarePower * 1000) / tputPerGpu` (Joules per token, where power in kW is converted to W). These divide one GPU's all-in power by that GPU's reported throughput; for disaggregated rows `output_tput_per_gpu` is per decode GPU, so `jOutput` ignores the prefill pool. Its semantics are frozen; the whole-deployment view lives in the power boundaries below. +- Power-boundary fields — see [Power boundaries](#power-boundaries). + +#### Power boundaries + +`lib/power-basis.ts` derives the three boundaries beyond GPU-measured telemetry as W per allocated GPU plus J per successful output token. `buildDerivedChartFields` calls `buildPowerBasisChartFields(entry, specs)` right after the measured fields, so the official chart, the `?unofficialrun=` overlay (`transformBenchmarkRows`), and Historical Trends (`rowToLightweightPoint` with `requestedMetrics`) share one formula set; each key is gated by `wants(key)` so selective output stays exact. + +| Basis | W / GPU | J / output token | Fields | +| --------------------------------------------- | -------------------------------------------------------------------------- | ---------------------------------- | ------------------------------------------------------------------ | +| B1 GPU measured | `avg_power_w` | `joules_per_output_token` | existing `measuredAvgPower`, `measuredJPerOutputToken` (unchanged) | +| B2 GPU provisioned | `HW_REGISTRY.tdp` | `W × N_alloc ÷ total output tok/s` | `gpuProvisionedWatts`, `gpuProvisionedJPerOutputToken` | +| B3 utility provisioned (all-in) | `HW_REGISTRY.power × 1000` | `W × N_alloc ÷ total output tok/s` | `utilityProvisionedWatts`, `utilityProvisionedJPerOutputToken` | +| B4 utility modeled (measured → chassis → PUE) | `modeledSystemPower.deploymentFacilityWatts ÷ modeledSystemPower.gpuCount` | `B1 J/out × (B4 W ÷ B1 W)` | `utilityModeledWatts`, `utilityModeledJPerOutputToken` | + +- **N_alloc and total throughput** (`powerBasisNormalization`). Aggregate rows report output per allocated GPU, so `N_alloc` cancels and the builder uses `W ÷ output_tput_per_gpu` without trusting display counts (legacy ingest can encode TP × EP twice). Fixed-sequence disaggregated rows (`disagg && benchmark_type === 'single_turn'`) report output per decode GPU while the deployment also powers the prefill pool, so `total output tok/s = output_tput_per_gpu × num_decode_gpu` and `N_alloc = num_prefill_gpu + num_decode_gpu`; the article's 4P+4D counts eight GPUs, and B3 energy is `(P + D) / D` times `jOutput` on such rows. Other disaggregated benchmark types emit no provisioned energy because it is not verifiable in-app whether AgentX throughput already divides by all GPUs. +- **B4 source.** Reuses the `SystemPowerEstimate` that `rowToAggDataEntry` attached as `entry.modeledSystemPower`; nothing re-runs `modelSystemPower`. `deploymentFacilityWatts` is chassis AC × PUE with PUE applied exactly once inside `estimateChassisPower`, divided by the physical measured GPU count (not `modeledGpuCount`, which over-counts partially allocated chassis: a 4P+4D deployment on two worker hosts models 16 GPUs while measuring 8, and the unit test pins that divisor). This is a different quantity from the existing `modeledChassisPowerPerGpu` (chassis AC ÷ modeled GPU count, no PUE). Modeled energy scales the producer's same-window `joules_per_output_token` by `B4 W ÷ avg_power_w`, so it inherits B1's token denominator and survives rows whose `output_tput_per_gpu` is missing (the provisioned energies do not; a per-basis point count cannot assume one shared denominator). +- **Null rules.** A value is emitted only when finite and positive; otherwise the key is omitted (never `{ y: 0 }`), because the metric filters drop points by `metricKey in point` and `remapInferencePoint` falls back to raw throughput when a key exists with an unusable value. B2/B3 W are absent for hardware without registry specs (`getGpuSpecs` returns zeros); their energies are also absent without output throughput or disaggregated counts. B4 requires `modeledSystemPower.status === 'supported'` and B1 watts (`avg_power_w` after `rowToAggDataEntry`'s `power_valid !== 0` admission); B4 energy additionally needs `joules_per_output_token`. Telemetry admission belongs to `modelSystemPower`, which accepts `power_valid === 1` with schema v2 or the validated unversioned single-node producer (`telemetryBasis: 'validated-unversioned-single-node'`), so B4 renders on exactly the rows that show B1 and `modeledChassisPowerPerGpu`; off-8k/1k workloads, unsupported hardware such as GB200/GB300 NVL72 (`reason: 'hardware'`), and telemetry/topology failures withhold it. The public API's `strictV2` row filter is not re-applied in the chart for B1, so it is not re-applied for B4 either (plan §3.1 describes B1 with that filter; the app's chart path is the authority here). +- **Ordering invariant** (unit-tested on real B200 telemetry): B3 ≥ B4 ≥ B1 and B2 ≥ B1 for both W/GPU and J/out on the same point. +- **Reconstructed prefill energy** (`reconstructedPrefillJPerOutputToken`, `utils/role-energy.ts`). For validated (`power_valid === 1`, schema 2) disaggregated rows, `prefill_joules_per_input_token × (joules_per_output_token ÷ joules_per_input_token)` carries the prefill pool's energy onto the output-token axis: the ratio is the served input:output token count because schema-2 aggregate energy has one numerator. With `decode_joules_per_output_token` it sums back to the deployment's J/out. It feeds only the `i_pcompare=roles` comparison on the energy axis (PowerX Figure 7) and is never a y-axis of its own; aggregate rows and rows missing any of the four inputs omit it. +- Historical Trends substitutes `output_tput_per_gpu := tput_per_gpu` for legacy rows lacking output throughput; provisioned energies in trends inherit that fallback. **GPU specs lookup** happens inside `buildDerivedChartFields`. `getGpuSpecs(hwKey)` (`lib/constants.ts`) splits on `[-_]` to extract the base GPU token (for example, `"b200_trt_mtp"` becomes `"b200"`) and looks it up in `HW_REGISTRY`. Missing keys return zeroed specs, producing `0` cost/energy values rather than crashing. diff --git a/docs/powerx-permanent-view.md b/docs/powerx-permanent-view.md new file mode 100644 index 000000000..8e3725ca2 --- /dev/null +++ b/docs/powerx-permanent-view.md @@ -0,0 +1,286 @@ +# PowerX Permanent View + +How the PowerX article's power-boundary figures live inside `/inference` instead of a +separate page. The measured-power UI, the four power boundaries, and their share links all +sit in the existing ↑↑↓↓-gated **Measured Energy** metric group. + +## Purpose + +The article compares one deployment's power at four boundaries, from the GPU board to the +utility meter, both as W per GPU and as J per output token. The dashboard already had +GPU-measured telemetry (`measured*` metrics) and an ungated all-in provisioned energy +(`jOutput`). This view adds the other boundaries as ordinary y-axis metrics so every chart +feature (zoom, tooltip, Perf Ruler, Optimal Only, Table, CSV, PNG, `?unofficialrun=` overlay, +share URL) applies unchanged. One chart engine, no second chart component. + +## Gate + +The Measured Energy group is `gated: true` in `metric-registry.ts`. `ChartControls` hides a +gated group while the gate is locked unless the selected `i_metric` belongs to it, so a +shared link to any boundary renders for a reader who never unlocked the gate, while the +group stays out of the selector otherwise. The boundary metrics are members of that group +(`POWER_BASIS_METRIC_CONFIG_KEYS` is spread into it) so they inherit exactly this behaviour. + +## Boundaries + +| Basis (`PowerBasis`) | Selector label | W / GPU metric | J / output token metric | Source | +| --------------------- | ---------------------------- | -------------------------------------------- | ------------------------------------------------- | ----------------------------------------------------------------------- | +| `gpu-measured` | GPU measured | `y_measuredAvgPower` (+P75/P90, roles, %TDP) | `y_measuredJPerOutputToken` (+ input/total/query) | runner telemetry; existing metrics, unchanged | +| `gpu-provisioned` | GPU provisioned (TDP) | `y_gpuProvisionedWatts` | `y_gpuProvisionedJPerOutputToken` | `HW_REGISTRY.tdp` | +| `utility-provisioned` | Utility provisioned (all-in) | `y_utilityProvisionedWatts` | `y_utilityProvisionedJPerOutputToken` | `HW_REGISTRY.power` (all-in kW per GPU) | +| `utility-modeled` | Utility modeled (PUE) | `y_utilityModeledWatts` | `y_utilityModeledJPerOutputToken` | `modelSystemPower` chassis AC × PUE 1.3 (applied once), ÷ measured GPUs | + +Formulas, the all-GPU normalization (`N_alloc` = prefill + decode GPUs for disaggregated +rows) and the null rules are specified in +[Data Transforms → Power boundaries](./data-transforms.md#power-boundaries); the chassis model +itself in [PowerX System Power](./powerx-system-power.md). The ungated `jOutput` keeps its +per-decode-GPU normalization; its labels and the boundary metric's `all GPUs` label keep the +two distinguishable in the selector, the availability list and CSV headers. + +Every boundary metric has `polarity: 'lower'`. Energy metrics therefore get the usual +lower-is-better Pareto frontier. The three watt keys are listed in `POWER_CURVE_METRICS` +(`utils/powerCurves.ts`), so `ScatterGraph`/`GPUGraph` draw the upper power envelope for them +exactly as for `y_measuredAvgPower`, and Optimal Only keeps every load point; the declared +`lower_*` roofline only drives the ascending Table sort. They stay out of +`isMeasuredPowerCurveMetric`, like `y_modeledChassisPowerPerGpu`, so telemetry-only decorations +do not apply. `measured-power-direction.test.ts` pins both halves. A flat provisioned series +(TDP or all-in W is one constant per hardware) has a single envelope vertex per hardware, so no +envelope line is drawn for it — only the points. + +## Why the metric key carries the boundary + +There is no `i_pbasis` URL parameter. The Boundary select in `MeasuredMetricControls` is +presentation over `selectedYAxisMetric`, exactly like Per / Scope / Statistic / Display / +Unit: `measured-metric-config.ts` gives every Measured Energy key a `basis` and resolves each +control change to the nearest registered key. Consequences: + +- One share link per figure, and `i_metric` already round-trips through every existing + surface (share button, Historical Trends hand-off, Table, CSV header). +- Existing links are byte-identical: all `measured*` keys carry `basis: 'gpu-measured'`, and + `MEASURED_METRIC_DEFAULTS` did not change. +- Choosing a derived boundary snaps to its canonical combination (whole deployment, average + watts, joules per output token). Changing any telemetry-only setting while on a derived + boundary returns to GPU measured, so no control ever points at a key that does not exist. +- A gated metric selected through a shared URL keeps its Measured controls, as the + `measured*` siblings already do. +- The Y-axis dropdown's collapsed "Measured Power" / "Measured Energy" option for the other + family resolves to `MEASURED_METRIC_DEFAULTS[family]` (GPU measured), so switching family + from the dropdown resets the boundary; the resolver itself keeps `basis` on a family change + (`changeMeasuredMetricConfig(key, { family })`), which is what a future family control in + the Measured row should call. + +The Boundary select tracks `inference_power_basis_changed { basis, family }` in addition to the +existing `inference_y_axis_metric_selected` fired by `ChartControls`. + +## Missing values + +A point without a value for the selected boundary is omitted from that series only (the +builders never emit `{ y: 0 }`, and both the official and overlay paths filter by +`metricKey in point`). The PowerX availability panel explains the gap per point: + +| State | Meaning | +| ------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `noSpec` | hardware has no `tdp` / `power` in `HW_REGISTRY` | +| `noThroughput` | provisioned watts exist but the row has no output throughput | +| `noNormalization` | disaggregated row with throughput but no whole-deployment GPU count (`powerBasisNormalization`: not `single_turn`, or non-integer prefill/decode counts) | +| `noTelemetry` | modeled boundary needs validated GPU telemetry (B4 follows B1) | +| `invalid` | `power_valid === 0` | +| `modelWorkload` | chassis model covers 8K / 1K only | +| `modelHardware` | hardware outside the chassis profiles (GB200 / GB300 NVL72) | +| `modelUnsupported` | other `modelSystemPower` reasons (gpu-count, topology, role-power …) | + +The chart caption (`data-testid="power-basis-assumptions"`) names the boundary, the formula in +words, the PUE constant and the chassis-model revision so a screenshot records its method. +The pinned tooltip's "Modeled system power" block (`tooltipUtils.ts` `modeledSystemPowerHTML`) +renders only for `measured*` keys and `y_modeledChassisPowerPerGpu`, so on the boundary keys the +caption is the only per-chart provenance; the caption does not promise more. +The empty state for the utility-modeled boundary says why nothing rendered and points at the +Boundary select. + +## Article figure → share link + +Base: `/inference?g_model=&i_seq=8k/1k&i_prec=` plus the hardware legend +selection. Append: + +| Figure | Content | Parameters | +| ------ | --------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| 2 | H200 W/GPU, all four boundaries on one chart | `&i_metric=y_measuredAvgPower&i_pcompare=boundaries` (one boundary at a time: `&i_metric=y_gpuProvisionedWatts` / `y_utilityProvisionedWatts` / `y_utilityModeledWatts`) | +| 3 | H200 J/output token, all four boundaries | `&i_metric=y_measuredJPerOutputToken&i_pcompare=boundaries` (single boundary: the matching `…JPerOutputToken` key) | +| 4 / 5 | B200 vs B300, GB200 vs GB300 at fixed X | `&i_metric=&i_rulers=` (Perf Ruler, T1) | +| 6 | Prefill vs decode W/GPU with the whole deploy | `&i_metric=y_measuredAvgPower&i_pcompare=roles` | +| 7 | Reconstructed request energy, prefill share | `&i_metric=y_measuredJPerOutputToken&i_pcompare=roles` (prefill share in the clone's tooltip) | +| 1 | Measured power over the benchmark job | `&i_metric=y_measuredPowerTimeline` (Display → Timeline, see below) | + +Pinned official dashboard points now offer **View PowerX** when the feature gate is unlocked +or a measured-metric share link is active. The lazy in-page dialog reads the existing +`GET /api/v1/gpu-metrics-point?id=N` API, without entering a run ID or changing chart filters. +This also works in the hardware/date comparison view. The per-chip chart and statistics span +the recorded job (including startup/warmup), not just the audited serving window. Fixed-sequence +points do not request AgentX server-metric overlays. Missing telemetry is shown as unavailable, +not zero power. Unofficial points have no database ID and retain **View power trace**, which +resolves their run/audit provenance automatically. No API contract or ingestion changes are +needed for this presentation-only entry point. + +The `/gpu-metrics` page keeps the raw per-run explorer (every metric, one artifact at a time); +the timeline below is the chart-scoped view of the same artifacts. + +## Comparison series (`i_pcompare`) + +`i_pcompare=boundaries|roles` (default `''`, "Off") overlays sibling series on the selected +metric without changing it. `utils/power-compare.ts` is the single source: `useChartData` and +the `?unofficialrun=` processor (`processOverlayChartDataWithClipping`) both call +`expandPowerCompareSeries`, which appends one clone per sibling to every base point — same +`x`, `y` taken from the sibling's field, `powerVariant` set — and leaves the base points +untouched, so a chart without a comparison is byte-identical to before. A point lacking a +sibling's value contributes nothing to that series (never a 0), so each boundary's and +role's availability rules carry through unchanged. + +| Mode | Base series | Siblings | Fields | +| ------------ | ------------------------------------------------- | -------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `boundaries` | the selected boundary (`basis` of the metric key) | the other three boundaries | `measuredAvgPower` / `POWER_BASIS_FIELDS[*].watts`, or `measuredJPerOutputToken` / `POWER_BASIS_FIELDS[*].energy` | +| `roles` | the selected scope (`all`, `prefill`, `decode`) | the other two worker pools | W: `measuredAvgPower`, `measuredPrefillAvgPower`, `measuredDecodeAvgPower`; J/out: `measuredJPerOutputToken`, `reconstructedPrefillJPerOutputToken`, `measuredDecodeJPerOutputToken` | + +Siblings exist only where the metric names the whole-deployment **average W/chip** or **J per +output token** (the only quantities every boundary and role publishes on one axis); role +energy additionally excludes the prefill J per input token scope. Elsewhere +`powerCompareVariants` returns nothing, the Compare select disables the option, and — when a +link arrives with an inapplicable mode — a hint (`measured-compare-hint`) says the comparison is +paused. The parameter is kept, so switching back to an applicable setting resumes it. + +Rendering (`ScatterGraph`): the series key is `scatterSeriesKey(point)` = +`_[-v-]` (`utils/point-identity.ts`), so rooflines, frontiers, +Optimal Only, line labels, overflow continuations and the perf ruler treat each sibling as its +own series; `parseScatterSeriesKey` recovers hardware, precision and variant wherever the key +was previously split on `_`. Siblings keep the hardware colour (overlay runs keep the run +colour) and take a per-variant `stroke-dasharray` (`powerVariantDash`); clone points render +at 0.6 opacity behind their base and carry `data-power-variant`. Line labels are placed per +hardware _and_ sibling: the base series keeps its plain hardware label, while a sibling appends +` · ` (`powerLineLabelSuffix`: TDP, All-in, PUE modeled, Prefill GPUs, …) and, when a +boundary is flat on a watts axis, its shared value (`B300 (SGLang) · TDP 1.2 kW`), so an exported +PNG explains its dashed lines without the legend; each pill carries `data-series-id` +(`::` for a sibling) and `data-power-variant`. The legend appends one +line-swatch row per series present (base first); rows toggle chart-local visibility +(`hiddenPowerVariants`, not in the URL) and hover-highlight that series across every hardware. +`scatterPointConfigId` includes the variant so a clone never replaces its base in a D3 join. +Tooltips add a "Series" line; on the energy axis a role clone also reports its share of the +reconstructed request energy. Table adds a "Series" column and CSV a trailing "Power Series" +column only while clones are present. Comparison clones are excluded from `bestSeriesPerSku`, +the power-tier counts, the legend points table, the availability panel (which reads +`selectionPoints`) and the date-comparison `GPUGraph`. Unofficial-run pills read `✕ ` (`getOverlayLineLabel`); the branch stays in the +legend and a short run tag (` · main`, ` · …-`) is appended only when several overlay +runs draw the same hardware. + +Figure 7's reconstruction lives in `utils/role-energy.ts`: schema-2 aggregate energy has one +numerator, so `J/out ÷ J/in` is the served input:output token ratio and +`prefill_joules_per_input_token × (J/out ÷ J/in)` is the prefill pool's energy per output +token; with `decode_joules_per_output_token` it sums back to the deployment's J/out when the +pool energies partition the total. `buildDerivedChartFields` emits it as +`reconstructedPrefillJPerOutputToken` for validated (`power_valid === 1`, schema 2) +disaggregated rows only; it is never a y-axis of its own. + +The Compare select tracks `inference_power_compare_changed { mode, family }`; legend rows +track `inference_power_compare_series_toggled { series, visible }`. + +## Power timeline (Display → Timeline) + +`y_measuredPowerTimeline` is the third value of the Measured Power **Display** control +(`watts` / `tdp` / `timeline`, `MeasuredPowerDisplay` in `measured-metric-config.ts`). Its +registry field aliases `measuredAvgPower`, so the point set, the availability panel, the Table +view and the share link are those of the measured average; only the chart body changes: +`ChartDisplay` renders `ui/PowerTimeline.tsx` instead of `ScatterGraph` when the resolved +config is `display: 'timeline'`. + +- **Join.** Each validated row's `power_audit.source` is `power_validation_.json`. + Two collectors publish the telemetry behind it (`utils/powerTimeline.ts`): + - single-node runners upload one `gpu_metrics_` CSV artifact per config + (`benchmark-tmpl.yml`), so the source names the artifact exactly + (`telemetryArtifactForPoint`); + - Slurm / Dynamo disaggregated runners upload one `power_audit_` bundle per + concurrency sweep whose `LOGS/power/samples.csv` (DCGM, `/` per device) + covers every concurrency. The API cuts it into one series per `power_validation_*.json` the + bundle contains (`components/gpu-power/power-audit-bundle.ts`: the validation file's + `selected_window` ± 60 s, roles from its `per_gpu_role`, manifest `expected_devices` as the + fallback) and labels each with that file name in `series.source`, which `joinPowerTimeline` + matches before falling back to the artifact name. + + Nothing is matched by hardware or concurrency. Rows whose telemetry is missing (expired + artifact, bundle over the download cap, another collector) are listed under the chart + (`data-testid="power-timeline-missing"`), never estimated. Trace keys are + `:` (`traceKeyForPoint`), unique per point in both collectors. + +- **Fetch.** One request per workflow run in the visible points + (`planPowerTimelineRequests`, at most `POWER_TIMELINE_MAX_RUNS`; a deep-linked trace's run + goes first, then `?unofficialrun=` overlay runs, then official runs — `prioritizeRun` / + `prioritizeRuns` — so an overlay the user asked for is never the run that gets dropped), + narrowed with + `prefix=` to the common RESULT_FILENAME prefix so a nightly sweep's other models are not + downloaded. `/api/gpu-metrics?series=power` first uses persisted telemetry and returns one-second per-GPU buckets + (`components/gpu-power/power-series.ts`, ~1 MB for a 25-config run instead of ~27 MB of + raw rows). Missing stored telemetry falls back to artifacts; database failures are explicit errors. + See [persistence and repair](./powerx-persistence-recovery.md) for cache freshness and coverage receipts. + With `series=power` the artifact fallback also downloads `power_audit_*` bundles whose name + shares the prefix (a bundle names the sweep, so the match runs both ways), reads only + `LOGS/power/samples.csv`, `LOGS/power/manifest.json` and the top-level + `power_validation_*.json` entries, and skips bundles above 256 MiB (the GB200 nw8 sweep is + 215 MB). A bundle-cut series carries `source` and `devices[] { id, role? }`; CSV series carry + neither. Runner timestamps are UTC wall clock and are parsed as such + (`parseTelemetryTimestampUtc`); `new Date('2026/09/12 20:19:57')` would read browser local + time and misplace the audit window. +- **Legend with overlays.** Once `?unofficialrun=` data is in, the chart reads + `localOfficialOverride`, so the timeline legend writes the unified selection + (`setUnifiedOverlaySelection`, `computeToggle` solo semantics) exactly as `ScatterGraph` + does; the context's `toggleHwType` alone would change nothing visible there. + +- **Drawing.** One trace per config, mean of its GPUs (legend switch: one line per GPU), + coloured by hardware for official rows and by `overlayRunColor(runIndex)` for + `?unofficialrun=` rows; legend toggles follow `activeHwTypes` / `activeOverlayHwTypes` + exactly as in `ScatterGraph`. The whole job is drawn faint and the `power_audit` window + emphasized (`data-segment="full" | "window"`). Rated TDP is a dashed reference per hardware; + the all-in provisioned line is an opt-in legend switch because it halves the traces' + vertical resolution. X axis: wall clock (UTC) when the visible traces come from one run, + otherwise seconds since each trace's start; both are a toolbar toggle. `c` labels sit + at the end of the emphasized segment. +- **Pools (Figure 1).** The legend switch _Prefill / decode pools_ (shown when a visible + trace carries worker roles) sums the board power of each role's GPUs + (`tracePools` / `sumPowerAt`) and draws one line per pool — prefill dashed `7 3`, decode + `2 3`, the same dashes as the `i_pcompare=roles` series — labelled `c · prefill` / + `· decode`, on a _GPU pool power (W)_ axis. Traces without roles draw their deployment total + as one `all` line so single-node configs stay comparable. Reference lines become pool size × + rated TDP per (hardware, pool size): roles of one hardware that hold the same GPU count share + one line labelled `prefill / decode ×16` (`data-pool` on `.power-reference` lists the roles, + `groupPoolsBySize`), and labels of lines at equal watts stack upward (`referenceLabelSlots`) + instead of overprinting; and the + all-in switch scales the same way. The tooltip names the pool, its summed watts against the + pool TDP, and the mean / min / max per GPU inside it. Pool mode and per-GPU lines are + mutually exclusive. +- **Deep link.** A pinned scatter tooltip on any metric of the measured family offers _View + power trace_ (`data-action="view-power-trace"`, official and `?unofficialrun=` points + alike). It switches the Display to Timeline in place and records the point's trace key in a + one-shot module store (`requestPowerTraceFocus`); the timeline consumes it on mount, dims + every other trace, switches to pool mode when the trace has roles, and shows a _Focused on …_ + chip (`data-testid="power-timeline-focus"`) with _Show all_ to clear. The link's `href` is + the current page with `i_metric=y_measuredPowerTimeline`, so open-in-new-tab lands on the + timeline (unfocused: the focus is a gesture, not URL state). +- **State.** Axis mode, line mode (mean / per GPU / pools), the all-in switch, the focused + trace and hover highlight are component state, not URL state: the share link is `i_metric=y_measuredPowerTimeline` plus the usual + scope, and a reader lands on the same defaults. +- **Analytics.** `inference_power_timeline_loaded { traces, missing, runs }`, + `inference_power_timeline_axis_changed { mode }`, + `inference_power_timeline_lines_changed { lines: 'mean' | 'gpu' | 'pool' }`, + `inference_power_timeline_utility_toggled { enabled }`, + `inference_power_trace_opened { hwKey, conc, overlay }` (scatter tooltip action), + `inference_power_timeline_focus_cleared`. + +## Tests + +- `lib/power-basis.test.ts` and `lib/chart-utils.test.ts` — the six boundary values through + the real builder, and the same fields on `?unofficialrun=` overlay rows. +- `gpu-power/power-series.test.ts`, `power-audit-bundle.test.ts` — one-second buckets with + `null` gaps, a partial pool is a gap rather than a lower sum, and bundle rows stay on their + host/device and role inside the padded window. +- `utils/powerTimeline.test.ts` — overlay runs are fetched ahead of official runs. +- `api/gpu-metrics/route*.test.ts` — the stored digest shape, DB-first serving, database + failure as 503, known missing hosts, and the retained-inventory recount before a CSV + fallback. +- `cypress/component/power-timeline.cy.tsx`, `power-compare.cy.tsx` — overlay-run colour and + the overlay hardware filter. diff --git a/docs/powerx-system-power.md b/docs/powerx-system-power.md index e1e1e5426..4eec8f70b 100644 --- a/docs/powerx-system-power.md +++ b/docs/powerx-system-power.md @@ -95,15 +95,20 @@ facility kW/GPU used to calculate capacity per GW. Consequently, revenue, compute expense, license fee, and profit scale together; profit margin does not change. Electricity expense is not recomputed separately. -This opt-in AgentX estimate requires validated schema-v2 telemetry and a -single-node chassis supported by the pinned model. Validated 1/2/4-GPU allocations -use the existing full-chassis extrapolation: fill an eight-GPU server with whole -replicas at the measured per-GPU power and throughput, then divide modeled facility -power by eight. This assumes replica co-location does not change performance or -power; it is not a measurement of a partly idle server. The chart, tooltip, and CSV -label every extrapolated estimate, including interpolation with one partial knot. -Unsupported GB200/GB300 chassis, multi-node layouts, allocations that cannot tile -eight GPUs, and missing/invalid measurements stay unavailable with distinct reasons. +This opt-in AgentX estimate requires validated schema-v2 telemetry and chassis +supported by the pinned model. Fully measured eight-GPU chassis are supported +on a single node, per measured worker host, or across an aggregate multinode +deployment without per-worker telemetry at the deployment-mean GPU power +(`topologyBasis: 'uniform-hosts'`; symmetric TP/PP/DP shards load each host alike). +Validated single-node 1/2/4-GPU allocations use full-chassis extrapolation: fill +an eight-GPU server with whole replicas at the measured per-GPU power and +throughput, then divide modeled facility power by eight. This assumes replica +co-location does not change performance or power; it is not a measurement of a +partly idle server. The chart, tooltip, and CSV label every extrapolated estimate, +including interpolation with one partial knot. Unsupported GB200/GB300 chassis, +partial multi-host allocations, disaggregated deployments without per-worker +telemetry, allocations that cannot tile eight GPUs, and missing/invalid +measurements stay unavailable with distinct reasons. The ordinary 8K/1K transformation keeps its existing admission policy. At an exact frontier point, use that point's modeled power. Between points, diff --git a/docs/state-ownership.md b/docs/state-ownership.md index d4922dae3..4c8ff69d5 100644 --- a/docs/state-ownership.md +++ b/docs/state-ownership.md @@ -113,12 +113,65 @@ them. Axis and presentation state: - selected x-axis and y-axis metrics, percentile, and effective x-axis mode +- the Measured controls (Boundary / Per / Scope / Statistic / Display / Unit) own no state: `measured-metric-config.ts` resolves each selection to the nearest registered metric key and writes it back to `selectedYAxisMetric`, so the power boundary rides on `i_metric` (there is deliberately no `i_pbasis`; see [PowerX Permanent View](./powerx-permanent-view.md)) +- the Measured Power Display value `timeline` (`y_measuredPowerTimeline`) swaps the chart body for `PowerTimeline`; its axis mode, per-GPU lines and all-in reference switch are component state and never enter the URL - token-revenue price source (`i_revenue`): normalized uncached/cached/output pricing or the selected model's live OpenRouter catalog prices - scale, optimal-point, label, contrast, legend, and overlay controls Display changes stay in this domain. A contrast or label toggle therefore does not notify filter-only or data-only consumers. +**`usePerfRulerStore`** (perf rulers, `i_rulers`) + +Completed Perf Rulers on the primary inference chart (`chart-0`) are provider state, +exposed through a dedicated `PerfRulerStoreContext` +(`packages/app/src/components/inference/perf-ruler-store.ts`) rather than the display domain: a +ruler commit would otherwise rerender every display consumer, and harnesses that mount +`InferenceContextsProvider` with static mock values would silently lose ruler +interactivity. The store is scoped to one chart id on purpose. The replay chart +(`replay-chart-0`) draws the same curve classes under the same provider, so a shared +store would render every ruler twice and let the replay's prune pass delete rulers the +main chart still shows; it and `GPUGraph` keep component-local state, which `ScatterGraph` +also falls back to when no store is present. + +`ScatterGraph` reads `state`/`setState` from the store, so the existing reducers, refs, +and draw passes are unchanged. The one thing that moved is the axis reset: +`usePerfRulerAxisReset` adjusts state during the chart's render, which is only legal for +the chart's own state, so for persisted rulers the same reset (either axis metric changes +→ rulers, draft, and pending share-link rulers are cleared) runs inside the provider. Its +key, `persistedPerfRulerAxisKey`, describes the graph `ChartDisplay` renders as `chart-0`, +not `graphs[0]`: `graphs` is always `[interactivity, e2e]` and ChartDisplay shows the e2e +graph for every non-interactivity x mode, so the key picks the graph by `selectedXAxisMode` +(as `bestHwTypes` does), uses its `x_scale_field` (which carries the percentile), and folds +the mode itself in because the derived agentic modes override the e2e graph's x field inside +ChartDisplay only. The first chart definition (null → key) and a reload of the same axes +(key → null → key) are not axis changes, so share-link rulers survive the load. + +Restoring from a link is a two-phase commit because of a load race. Rulers parsed from +`i_rulers` start as `pending`; data, `i_gpus`, comparison dates, and `?unofficialrun=` +overlays all arrive after the chart's first draw, and the chart prunes any committed +ruler whose curve path is absent from the DOM. The chart therefore commits a pending +ruler only once BOTH of its curve paths exist (hidden-at-opacity-0 counts as present, as +for prune), clamping the iso-x to the pair's overlap through the drawn paths. The commit +runs inside the perf-ruler D3 layer's draw pass (`drawPerfRuler`), not in a React effect: +the chart's first draw happens in a `D3Chart`-local re-render after its container is +measured, which re-renders nothing in `ScatterGraph`, so an effect keyed on the chart's +props could miss the first draw and leave resolvable rulers pending for the session. No +commit happens while the ruler mode is off (the power envelope forces it off on +non-measured axes; the mode-off effect discards pending rulers instead). Rulers whose +curves never appear stay pending — invisible and unserialized — until an axis change, +toggle-off, or clear discards them; this is a deliberate choice over a "data settled" +readiness predicate (fragile across the async order of availability, benchmark, overlay, +and derived-metric fetches). The visible cost is that a link whose curves the recipient's +view lacks lands with the ruler switch on and nothing drawn, and a later legend change on +the same axes can surface the ruler. A non-empty restore switches the ruler mode on for +that chart instance (rulers only render in mode) and fires +`interactivity_perf_ruler_shared_load` once per opened link, with `count` = the number of +rulers the link carried, the first time any of them renders — never per commit batch, so +the event counts links, not curve arrivals. Because the URL is read in a `useState` +initialiser, a retained-provider tab switch onto `/inference?i_rulers=…` does not +re-import the param — the same limitation as the other `i_*` toggles. + **`useInferenceActions`** All inference commands and setters. The context value and every exposed action keep @@ -360,3 +413,24 @@ Dashboard scope membership is declared by `shareParamScopes` in `packages/app/src/lib/dashboard-routes.ts`. Tests enforce completeness and route-specific share behavior, so this document deliberately does not duplicate a manually maintained parameter table. + +One entry needs a note on its encoding: `i_rulers` (inference scope, default `''`) is +`serializePerfRulers` output — `isoX|curveA|curveB` per ruler joined by `;`, where the +curve ids are the rendered roofline path identity classes (`roofline-_`, +`overlay-roofline-__run`, optionally `__`) and the +iso-x is in DATA space rounded to four significant digits. `parsePerfRulers` only accepts +curve ids of that identity-class shape — the shared `roofline-path` marker class or any +other zoom-group node would match many paths and draw a ruler between arbitrary curves. +Only committed rulers are serialized (never the draft or still-pending link rulers), and +the chart prunes them against the rendered curves, so a link written from one data set +degrades to fewer rulers, never to an error. Overlay ids depend on the run order in `unofficialruns`, which the share +link already carries. + +`i_pcompare` (inference scope, default `''`) is the power comparison mode, `boundaries` or +`roles`, owned by `InferenceProvider` as `powerCompare` (display context) with +`setPowerCompare` (actions). It is written as `''` for `none`. It is deliberately NOT cleared +when the metric changes to one without a common axis for the siblings: the chart then draws +the metric alone and the Measured controls show a hint, so the link's intent survives a +detour through another setting. Which comparison rows a reader hid from the legend is +`ScatterGraph` component state (`hiddenPowerVariants`), never shared — see +[PowerX Permanent View](./powerx-permanent-view.md#comparison-series-i_pcompare). From a72b9307a3e6a102d0390ff982e8aec46a4eb204 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Tue, 29 Sep 2026 12:56:23 -0700 Subject: [PATCH 6/7] fix: light the base series when its comparison legend row is hovered MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Base points and rooflines carry no power variant (only role and boundary siblings are cloned), while the base legend row's key holds the base id. Hovering that row therefore matched nothing and dimmed the whole chart. Map the empty variant back onto the base id for hover matching, and cover it with a component case that fails on the old code. 中文:基础系列的点和 roofline 不带 power variant(只有角色/边界的 兄弟系列会被克隆),而基础图例行的 key 带有基础 id,悬停时匹配不到 任何元素,整张图全部变暗。现将空 variant 映射回基础 id 用于悬停匹配, 并新增旧代码会失败的组件测试用例。 --- .../cypress/component/power-compare.cy.tsx | 29 +++++++++++++++++++ .../components/inference/ui/ScatterGraph.tsx | 10 ++++--- 2 files changed, 35 insertions(+), 4 deletions(-) diff --git a/packages/app/cypress/component/power-compare.cy.tsx b/packages/app/cypress/component/power-compare.cy.tsx index e1d3917a0..f2a51b485 100644 --- a/packages/app/cypress/component/power-compare.cy.tsx +++ b/packages/app/cypress/component/power-compare.cy.tsx @@ -151,4 +151,33 @@ describe('ScatterGraph power comparison series', () => { expect(hidden).to.have.length(3); }); }); + + it('keeps the base series lit when its legend row is hovered', () => { + const official = expandPowerCompareSeries(measuredCurve('b200'), 'y_measuredAvgPower', 'roles'); + const overlay = expandPowerCompareSeries( + measuredCurve('h100', OVERLAY_RUN_URL), + 'y_measuredAvgPower', + 'roles', + ); + mountCompare(official, overlay); + + cy.get(`${svg} .roofline-path[data-power-variant="prefill"]`).should('exist'); + // Base points and rooflines carry no variant id (only role siblings are + // cloned), so the base row has to map back onto them. + cy.get(legend).contains('label', 'All GPUs').trigger('mouseover'); + // Wait for the siblings to settle first: opacity transitions over 150 ms, + // so a base check taken at once would still read the pre-hover value. + cy.get(`${svg} .roofline-path[data-power-variant="prefill"]`).should( + 'have.css', + 'opacity', + '0.15', + ); + cy.get(`${svg} .roofline-path:not([data-power-variant])`).should('have.css', 'opacity', '1'); + cy.get(legend).contains('label', 'All GPUs').trigger('mouseout'); + cy.get(`${svg} .roofline-path[data-power-variant="prefill"]`).should( + 'have.css', + 'opacity', + '1', + ); + }); }); diff --git a/packages/app/src/components/inference/ui/ScatterGraph.tsx b/packages/app/src/components/inference/ui/ScatterGraph.tsx index 9f6656adf..318e2dce8 100644 --- a/packages/app/src/components/inference/ui/ScatterGraph.tsx +++ b/packages/app/src/components/inference/ui/ScatterGraph.tsx @@ -2396,14 +2396,16 @@ const ScatterGraph = React.memo( if (!svg) return; const root = d3.select(svg); // A comparison-series legend row highlights that boundary / role - // across every hardware instead of one hardware across series. + // across every hardware instead of one hardware across series. Base + // points and rooflines carry no variant (only siblings are cloned), + // so the empty id maps back to the base legend row's id. const variantId = hwKey.startsWith(POWER_VARIANT_LEGEND_PREFIX) ? hwKey.slice(POWER_VARIANT_LEGEND_PREFIX.length) : null; const matchesPoint = (d: InferenceData) => variantId === null ? String(d.hwKey) === hwKey - : powerVariantId(d.powerVariant) === variantId; + : (powerVariantId(d.powerVariant) || powerCompareBaseId) === variantId; root .selectAll('.dot-group') .style('opacity', (d) => @@ -2416,7 +2418,7 @@ const ScatterGraph = React.memo( const matches = variantId === null ? this.dataset.hwKey === hwKey - : (this.dataset.powerVariant ?? '') === variantId; + : (this.dataset.powerVariant || powerCompareBaseId) === variantId; return matches ? null : '0.15'; }); root @@ -2425,7 +2427,7 @@ const ScatterGraph = React.memo( return labelOpacityForHover((this as SVGGElement).dataset, hwKey); }); }, - [isPointVisible, isRooflineVisible], + [isPointVisible, isRooflineVisible, powerCompareBaseId], ); const handleLegendHoverEnd = useCallback(() => { From d10cdd0cb43d4ea4eb0eb983d4f2815b26a3ad77 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Tue, 29 Sep 2026 12:56:33 -0700 Subject: [PATCH 7/7] fix: memoise timeline responses on a fixed-shape input MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The response map spread one query result per run into its useMemo deps, so the array changed length whenever runs joined or left the plot and React logged the changed-size warning and recomputed. Build the resolved list in useQueries' combine, which React Query structurally shares, and memoise the map on that single input. Loading count and errors move into the same combine. A component case grows the plot from one run to two and asserts the warning is gone. 中文:响应 map 之前把每个 run 的查询结果展开进 useMemo 依赖数组, run 加入或离开图表时数组长度变化,React 会报 changed-size 警告并 重新计算。现改为在 useQueries 的 combine 中构建已解析列表(React Query 会做结构共享),map 只依赖这一个输入;加载计数与错误也移入同一 combine。 新增组件用例:图表从一个 run 增长到两个并断言不再出现该警告。 --- .../cypress/component/power-timeline.cy.tsx | 84 ++++++++++++++++--- .../components/inference/ui/PowerTimeline.tsx | 41 ++++----- 2 files changed, 94 insertions(+), 31 deletions(-) diff --git a/packages/app/cypress/component/power-timeline.cy.tsx b/packages/app/cypress/component/power-timeline.cy.tsx index 01fc78f13..7df453db1 100644 --- a/packages/app/cypress/component/power-timeline.cy.tsx +++ b/packages/app/cypress/component/power-timeline.cy.tsx @@ -1,4 +1,5 @@ import { PathnameContext } from 'next/dist/shared/lib/hooks-client-context.shared-runtime'; +import { useState } from 'react'; import type { GpuPowerSeries, GpuPowerSeriesResponse } from '@/components/gpu-power/power-series'; import PowerTimeline from '@/components/inference/ui/PowerTimeline'; @@ -21,6 +22,8 @@ const RUN_ID = '34716669498'; const RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${RUN_ID}`; const OVERLAY_RUN_ID = '31415926535'; const OVERLAY_RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${OVERLAY_RUN_ID}`; +const SECOND_RUN_ID = '34716669499'; +const SECOND_RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${SECOND_RUN_ID}`; const START_MS = Date.UTC(2026, 8, 12, 20, 20, 0); const hwConfig = createMockHardwareConfig(); const HW_TYPES = new Set(['b200', 'h100']); @@ -84,6 +87,24 @@ const response: GpuPowerSeriesResponse = { series: [series('b200', 16, 700), series('b200', 64, 900)], }; +const Y_LABEL = 'Measured Average Power per Chip over Time (W)'; + +function providerOverrides( + unofficial: Parameters[0] = {}, +): Parameters[1] { + return { + inference: { + selectedModel: Model.DeepSeek_V4_Pro, + selectedSequence: Sequence.EightK_OneK, + selectedYAxisMetric: 'y_measuredPowerTimeline', + hardwareConfig: hwConfig, + activeHwTypes: new Set(HW_TYPES), + hwTypesWithData: new Set(HW_TYPES), + }, + unofficial, + }; +} + function mountTimeline( data: InferenceData[], options: { @@ -98,21 +119,32 @@ function mountTimeline( chartId="power-timeline-test" data={data} overlayData={options.overlay} - yLabel="Measured Average Power per Chip over Time (W)" + yLabel={Y_LABEL} /> , - { - inference: { - selectedModel: Model.DeepSeek_V4_Pro, - selectedSequence: Sequence.EightK_OneK, - selectedYAxisMetric: 'y_measuredPowerTimeline', - hardwareConfig: hwConfig, - activeHwTypes: new Set(HW_TYPES), - hwTypesWithData: new Set(HW_TYPES), - }, - unofficial: options.unofficial ?? {}, - }, + providerOverrides(options.unofficial), + ); +} + +/** One measured run at mount; a button adds a second run to the same plot. */ +function GrowingTimeline() { + const [data, setData] = useState(() => [measuredPoint('b200', 16, 700)]); + return ( + + +
    + +
    +
    ); } @@ -181,4 +213,32 @@ describe('PowerTimeline', () => { }); cy.get('[data-testid="chart-legend"]').should('contain.text', '✕ powerx-timeline'); }); + + it('joins a run that arrives after mount without a hook-shape warning', () => { + const secondResponse: GpuPowerSeriesResponse = { + runInfo: { ...response.runInfo, id: Number(SECOND_RUN_ID), url: SECOND_RUN_URL }, + series: [series('h100', 16, 500)], + }; + cy.intercept('POST', `/api/gpu-metrics?runId=${RUN_ID}*`, { body: response }).as('first'); + cy.intercept('POST', `/api/gpu-metrics?runId=${SECOND_RUN_ID}*`, { + body: secondResponse, + }).as('second'); + cy.stub(console, 'error').as('consoleError'); + mountWithProviders(, providerOverrides()); + cy.wait('@first'); + svg().find('path.power-trace[data-segment="window"]').should('have.length', 1); + + cy.get('[data-testid="add-run"]').click(); + cy.wait('@second'); + svg().find('path.power-trace[data-segment="window"]').should('have.length', 2); + // React logs this when a memo's dependency array changes length between + // renders; one query per run used to be spread into that array. + cy.get('@consoleError').then((stub) => { + const calls = (stub as unknown as { args: unknown[][] }).args; + const shapeWarnings = calls.filter((args) => + args.some((a) => typeof a === 'string' && a.includes('changed size between renders')), + ); + expect(shapeWarnings, JSON.stringify(shapeWarnings)).to.have.length(0); + }); + }); }); diff --git a/packages/app/src/components/inference/ui/PowerTimeline.tsx b/packages/app/src/components/inference/ui/PowerTimeline.tsx index 6a56f0c4d..45ccecf58 100644 --- a/packages/app/src/components/inference/ui/PowerTimeline.tsx +++ b/packages/app/src/components/inference/ui/PowerTimeline.tsx @@ -24,7 +24,7 @@ * here focused on one config (`requestPowerTraceFocus`). */ import * as d3 from 'd3'; -import { useQueries } from '@tanstack/react-query'; +import { useQueries, type UseQueryResult } from '@tanstack/react-query'; import React, { useCallback, useEffect, useMemo, useRef, useState } from 'react'; import { HW_REGISTRY } from '@semianalysisai/inferencex-constants'; @@ -761,7 +761,25 @@ export default function PowerTimeline({ [requests, overlayRunIds], ); const droppedRuns = requests.length - fetchedRequests.length; - const queries = useQueries({ + // React Query structurally shares the combined result, so `resolved` keeps + // its identity until a run's data actually changes. That gives the response + // map a fixed-shape memo input; a per-query spread would change the deps + // array length whenever runs enter or leave the plot, which React rejects. + const combineQueries = useCallback( + (results: UseQueryResult[]) => ({ + loadingRuns: results.filter((query) => query.isPending).length, + errors: fetchedRequests.flatMap((request, index) => { + const error = results[index]?.error; + return error instanceof Error ? [{ request, error }] : []; + }), + resolved: fetchedRequests.flatMap((request, index) => { + const response = results[index]?.data; + return response ? [[request.runId, response] as const] : []; + }), + }), + [fetchedRequests], + ); + const { loadingRuns, errors, resolved } = useQueries({ queries: fetchedRequests.map((request) => ({ queryKey: ['power-timeline', request.runId, request.prefix, request.sources] as const, queryFn: ({ signal }: { signal: AbortSignal }) => fetchPowerSeries(request, signal), @@ -769,24 +787,9 @@ export default function PowerTimeline({ refetchOnWindowFocus: true, retry: 1, })), + combine: combineQueries, }); - const loadingRuns = queries.filter((query) => query.isPending).length; - const errors = fetchedRequests - .map((request, index) => ({ request, error: queries[index].error })) - .filter( - (entry): entry is { request: PowerTimelineRequest; error: Error } => - entry.error instanceof Error, - ); - const responses = useMemo(() => { - const map = new Map(); - fetchedRequests.forEach((request, index) => { - const response = queries[index].data; - if (response) map.set(request.runId, response); - }); - return map; - // Query data objects are stable per fetch; deriving from them keeps the map memoised. - // eslint-disable-next-line react-hooks/exhaustive-deps - }, [fetchedRequests, ...queries.map((query) => query.data)]); + const responses = useMemo(() => new Map(resolved), [resolved]); const { traces, missing } = useMemo( () => joinPowerTimeline(allPoints, responses),