From d5e90c4047ee5f04492f08491e59c198ca11ef91 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 17 Sep 2026 12:29:07 -0500 Subject: [PATCH 001/103] feat(powerx): digest gpu_metrics telemetry into the database at ingest time PowerX read chip telemetry by downloading and parsing gpu_metrics GitHub artifacts on every page view, and lost the data once GitHub's 90-day artifact retention expired. This moves that work to ingest time. - Migration 016 adds gpu_metric_series (one row per artifact CSV), gpu_metric_samples (full-resolution per-GPU samples), gpu_metric_gpu_stats (per-GPU min/max/mean/median/p95/p99/stddev digest), and benchmark_result_gpu_metrics (point <-> series links). - A pure nvidia-smi/amd-smi CSV parser plus artifact discovery and an idempotent upsert (same CSV hash refreshes links only; a changed CSV replaces samples and digest in one transaction). - CI ingest links gpu_metrics_ next to bmk_ using the same pairing rule as server logs. - New backfill CLI: bun run admin:db:backfill-gpu-metrics --all --yes (bounded by GitHub retention; the GCS backup does not mirror gpu_metrics). - /api/gpu-metrics serves the stored digest first and falls back to live GitHub artifacts for runs that are not ingested yet. - New /api/v1/gpu-metrics-point?id=N and a PowerX tab on the per-point detail page showing the telemetry recorded while that point ran. - Shared benchmark-result lookup extracted from the server-log backfill. Co-Authored-By: Claude Fable 5.1 --- AGENTS.md | 1 + docs/data-pipeline.md | 40 ++ package.json | 1 + .../app/src/app/api/gpu-metrics/route.test.ts | 223 +++++++++- packages/app/src/app/api/gpu-metrics/route.ts | 96 ++++- .../src/app/api/v1/gpu-metrics-point/route.ts | 39 ++ .../agentic-point/agentic-point-detail.tsx | 6 + .../agentic-point/power-telemetry-view.tsx | 255 ++++++++++++ .../agentic-point/use-detail-view.test.tsx | 22 + .../agentic-point/use-detail-view.ts | 8 +- .../src/hooks/api/use-gpu-metrics-point.ts | 18 + packages/constants/src/tables.ts | 4 + packages/db/migrations/016_gpu_metrics.sql | 99 +++++ packages/db/package.json | 1 + packages/db/src/backfill-gpu-metrics.ts | 303 ++++++++++++++ packages/db/src/backfill-server-log-files.ts | 97 +---- .../db/src/etl/gpu-metrics-artifacts.test.ts | 98 +++++ packages/db/src/etl/gpu-metrics-artifacts.ts | 173 ++++++++ packages/db/src/etl/gpu-metrics-csv.test.ts | 143 +++++++ packages/db/src/etl/gpu-metrics-csv.ts | 385 ++++++++++++++++++ .../db/src/etl/gpu-metrics-ingest.test.ts | 203 +++++++++ packages/db/src/etl/gpu-metrics-ingest.ts | 297 ++++++++++++++ packages/db/src/ingest-ci-run.ts | 35 ++ .../db/src/lib/benchmark-result-lookup.ts | 97 +++++ .../db/src/lib/gpu-metrics-backfill.test.ts | 49 +++ packages/db/src/lib/gpu-metrics-backfill.ts | 47 +++ packages/db/src/queries/gpu-metrics.ts | 342 ++++++++++++++++ packages/db/src/verify-db.ts | 3 + 28 files changed, 2987 insertions(+), 98 deletions(-) create mode 100644 packages/app/src/app/api/v1/gpu-metrics-point/route.ts create mode 100644 packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx create mode 100644 packages/app/src/hooks/api/use-gpu-metrics-point.ts create mode 100644 packages/db/migrations/016_gpu_metrics.sql create mode 100644 packages/db/src/backfill-gpu-metrics.ts create mode 100644 packages/db/src/etl/gpu-metrics-artifacts.test.ts create mode 100644 packages/db/src/etl/gpu-metrics-artifacts.ts create mode 100644 packages/db/src/etl/gpu-metrics-csv.test.ts create mode 100644 packages/db/src/etl/gpu-metrics-csv.ts create mode 100644 packages/db/src/etl/gpu-metrics-ingest.test.ts create mode 100644 packages/db/src/etl/gpu-metrics-ingest.ts create mode 100644 packages/db/src/lib/benchmark-result-lookup.ts create mode 100644 packages/db/src/lib/gpu-metrics-backfill.test.ts create mode 100644 packages/db/src/lib/gpu-metrics-backfill.ts create mode 100644 packages/db/src/queries/gpu-metrics.ts diff --git a/AGENTS.md b/AGENTS.md index f9bd2c868..ad909b256 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -76,6 +76,7 @@ API routes (`packages/app/src/app/api/v1/`): - `reliability` — raw `ReliabilityRow[]` - `evaluations` — raw `EvalRow[]` - `server-log` — retrieve benchmark runtime logs +- `gpu-metrics-point?id=N` — PowerX chip telemetry (samples + per-GPU digest) linked to one benchmark point - `invalidate` — invalidate API cache (admin; `?scope=collectivex` purges only that scope) - `collectivex/latest`, `collectivex/runs`, `collectivex/runs/[runId]` — CollectiveX sweep data from a **separate** Neon DB, populated lazily on read from GitHub Actions artifacts and served diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index 89b532996..ebafd3005 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -523,6 +523,46 @@ Producers (`aggregate_power.py`) annotate every aggregate result row with two op Reads are **permanently tolerant**: `queries/benchmarks.ts` selects the columns as `to_jsonb(br) -> 'power_invalid_reasons'` (and `lb` on the matview branch) rather than bare column references. A bare reference fails during query planning until the next ingest workflow applies the migration, because migrations run in the ingest workflows rather than at Vercel deploy. The key lookup degrades to NULL while the column is missing and is byte-identical once it exists, making deploy order irrelevant. +### PowerX Telemetry Digest (`gpu_metric_*`, migration 016) + +Every benchmark job samples `nvidia-smi` / `amd-smi` once per second for its +whole lifetime and uploads the CSV as `gpu_metrics_` next to +`bmk_` (agentic jobs: `bmk_agentic_`, still paired by the bare +suffix). The PowerX explorer used to download and parse those artifacts from +GitHub on every request and lost them after GitHub's 90-day retention. CI ingest +now digests them at ingest time, in the same step that links server logs: + +- `gpu_metric_series` — one row per (workflow run, artifact, CSV path): vendor, + CSV sha256, sample count, GPU count, recorded window, median cadence, and the + parsed sidecars (`gpu_metrics_context.json`, identity, amd-smi energy counters). +- `gpu_metric_samples` — full-resolution rows, one per (GPU, sample). NVIDIA + fills the six common columns; AMD additionally fills edge/memory temperature, + voltages, FCLK/SOCCLK and multimedia activity. Timestamps are UTC; NVIDIA's + zone-less `YYYY/MM/DD HH:MM:SS.mmm` is interpreted with the context sidecar's + `timestamp_timezone` (the producer writes UTC). +- `gpu_metric_gpu_stats` — per (series, GPU, metric) count/min/max/mean/median/ + p95/p99/stddev computed once at ingest so readers never rescan samples. +- `benchmark_result_gpu_metrics` — links each benchmark point to the series that + was recorded while it ran (several series per point for multinode artifacts). + +Ingest is idempotent: the same CSV hash refreshes only the point links, a changed +CSV replaces the samples and digest inside one transaction, and repeated final +samples (the monitor's stop-time flush) collapse on the primary key. Series are +stored per artifact, not per point: an AgentX per-concurrency job maps to one +point, while older fixed-sequence jobs that swept several concurrencies in one +job share one series across points. Windowing a series to the measured serving +interval is a reader concern; the raw series deliberately includes server +start-up and warm-up so both phases can be inspected. + +`bun run admin:db:backfill-gpu-metrics --all --yes` attaches telemetry for runs +ingested before this migration. The GCS backup only mirrors `bmk_`/`server_logs_` +uploads, so the reachable history is bounded by GitHub's 90-day retention. + +Readers: `/api/gpu-metrics?runId=` serves the digest when the run is stored and +falls back to live GitHub artifacts otherwise (in-progress runs), and +`/api/v1/gpu-metrics-point?id=` powers the PowerX tab of the per-point detail +page. + ### PowerX publication receipts The normal CI importer writes `POWER_PUBLICATION_MANIFEST` when configured. Each diff --git a/package.json b/package.json index d21c65db5..45a873cea 100644 --- a/package.json +++ b/package.json @@ -44,6 +44,7 @@ "admin:db:migrate:collectivex": "bun run --cwd packages/db db:migrate:collectivex", "admin:db:apply-overrides": "bun run --cwd packages/db db:apply-overrides", "admin:db:backfill-full-response-interactivity": "bun run --cwd packages/db db:backfill-full-response-interactivity", + "admin:db:backfill-gpu-metrics": "bun run --cwd packages/db db:backfill-gpu-metrics", "admin:db:backfill-server-log-files": "bun run --cwd packages/db db:backfill-server-log-files", "admin:db:reset": "bun run --cwd packages/db db:reset", "admin:db:verify": "bun run --cwd packages/db db:verify", diff --git a/packages/app/src/app/api/gpu-metrics/route.test.ts b/packages/app/src/app/api/gpu-metrics/route.test.ts index c34a2147f..790a0ff64 100644 --- a/packages/app/src/app/api/gpu-metrics/route.test.ts +++ b/packages/app/src/app/api/gpu-metrics/route.test.ts @@ -28,6 +28,18 @@ vi.mock('@/components/gpu-power/types', () => ({ parseCsvData: mockParseCsvData, })); +const { mockGetGpuMetricsForRun } = vi.hoisted(() => ({ + mockGetGpuMetricsForRun: vi.fn(), +})); + +vi.mock('@semianalysisai/inferencex-db/connection', () => ({ + getDb: () => ({}), +})); + +vi.mock('@semianalysisai/inferencex-db/queries/gpu-metrics', () => ({ + getGpuMetricsForRun: mockGetGpuMetricsForRun, +})); + vi.mock('adm-zip', () => { const csvContent = 'timestamp,index,power\n2026-03-01T00:00:00Z,0,300'; class MockAdmZip { @@ -44,11 +56,12 @@ vi.mock('adm-zip', () => { return { default: MockAdmZip }; }); -import { GET } from './route'; +import { databasePayloadToResponse, GET } from './route'; import { NextRequest } from 'next/server'; const originalFetch = globalThis.fetch; let origToken: string | undefined; +let origReadonlyUrl: string | undefined; function req(url: string): NextRequest { return new NextRequest(new URL(url, 'http://localhost')); @@ -57,7 +70,11 @@ function req(url: string): NextRequest { beforeEach(() => { vi.clearAllMocks(); origToken = process.env.GITHUB_TOKEN; + origReadonlyUrl = process.env.DATABASE_READONLY_URL; process.env.GITHUB_TOKEN = 'test-gh-token'; + // No readonly URL: the GitHub fallback is exercised unless a test opts in. + delete process.env.DATABASE_READONLY_URL; + mockGetGpuMetricsForRun.mockResolvedValue(null); }); afterEach(() => { @@ -67,6 +84,210 @@ afterEach(() => { } else { process.env.GITHUB_TOKEN = origToken; } + if (origReadonlyUrl === undefined) { + delete process.env.DATABASE_READONLY_URL; + } else { + process.env.DATABASE_READONLY_URL = origReadonlyUrl; + } +}); + +const storedRunPayload = { + workflowRun: { + id: 7, + githubRunId: 34557177019, + runAttempt: 1, + name: 'Run Sweep - dsr1 fp4 b200', + date: '2026-09-11', + htmlUrl: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34557177019', + headBranch: 'main', + headSha: 'deadbeef', + conclusion: 'success', + status: 'completed', + createdAt: '2026-09-11T04:00:00.000Z', + }, + series: [ + { + id: 1, + artifactName: 'gpu_metrics_dsr1_conc32_b200-x_0', + configKey: 'dsr1_conc32_b200-x_0', + fileName: 'gpu_metrics.csv', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 2, + gpuCount: 1, + startedAt: '2026-09-11T04:19:41.982Z', + endedAt: '2026-09-11T04:19:42.990Z', + sidecars: {}, + benchmarkResultIds: [10], + stats: [], + data: [ + { + timestamp: '2026-09-11T04:19:41.982Z', + index: 0, + power: 187.8, + temperature: 33, + smClock: 120, + memClock: 3996, + gpuUtil: 0, + memUtil: 0, + }, + { + timestamp: '2026-09-11T04:19:42.990Z', + index: 0, + power: 912.1, + temperature: 61, + smClock: 1965, + memClock: 3996, + gpuUtil: 98, + memUtil: 74, + }, + ], + }, + { + id: 2, + artifactName: 'gpu_metrics_multinode_b200-x_0', + configKey: 'multinode_b200-x_0', + fileName: 'results/gpu_metrics_rank0.csv', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 0, + gpuCount: 0, + startedAt: '2026-09-11T04:19:41.982Z', + endedAt: '2026-09-11T04:19:41.982Z', + sidecars: {}, + benchmarkResultIds: [11], + stats: [], + data: [], + }, + { + id: 3, + artifactName: 'gpu_metrics_multinode_b200-x_0', + configKey: 'multinode_b200-x_0', + fileName: 'results/gpu_metrics_rank1.csv', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 0, + gpuCount: 0, + startedAt: '2026-09-11T04:19:41.982Z', + endedAt: '2026-09-11T04:19:41.982Z', + sidecars: {}, + benchmarkResultIds: [11], + stats: [], + data: [], + }, + ], +}; + +describe('databasePayloadToResponse', () => { + it('shapes the stored digest like the GitHub payload and disambiguates multinode CSVs', () => { + const response = databasePayloadToResponse(storedRunPayload); + expect(response.source).toBe('database'); + expect(response.runInfo).toEqual({ + id: 34557177019, + name: 'Run Sweep - dsr1 fp4 b200', + branch: 'main', + sha: 'deadbeef', + createdAt: '2026-09-11T04:00:00.000Z', + url: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34557177019', + conclusion: 'success', + status: 'completed', + }); + expect(response.artifacts.map((artifact) => artifact.name)).toEqual([ + 'gpu_metrics_dsr1_conc32_b200-x_0', + 'gpu_metrics_multinode_b200-x_0/results/gpu_metrics_rank0.csv', + 'gpu_metrics_multinode_b200-x_0/results/gpu_metrics_rank1.csv', + ]); + expect(response.artifacts[0]!.data).toHaveLength(2); + expect(response.artifacts[0]!.series?.benchmarkResultIds).toEqual([10]); + expect(response.artifacts[0]!.series).not.toHaveProperty('data'); + }); + + it('fills missing run metadata with the run date and canonical run URL', () => { + const response = databasePayloadToResponse({ + ...storedRunPayload, + workflowRun: { + ...storedRunPayload.workflowRun, + htmlUrl: null, + headBranch: null, + headSha: null, + conclusion: null, + status: null, + createdAt: null, + }, + }); + expect(response.runInfo.createdAt).toBe('2026-09-11T00:00:00Z'); + expect(response.runInfo.url).toBe( + 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34557177019', + ); + expect(response.runInfo.branch).toBe(''); + }); +}); + +describe('GET /api/gpu-metrics — database first', () => { + it('serves the stored digest without touching GitHub when the run is ingested', async () => { + process.env.DATABASE_READONLY_URL = 'postgresql://readonly.example.test/db'; + mockGetGpuMetricsForRun.mockResolvedValueOnce(storedRunPayload); + globalThis.fetch = vi.fn(); + + const res = await GET(req('/api/gpu-metrics?runId=34557177019')); + expect(res.status).toBe(200); + const body = await res.json(); + expect(body.source).toBe('database'); + expect(body.artifacts).toHaveLength(3); + expect(mockGetGpuMetricsForRun).toHaveBeenCalledWith({}, 34557177019); + expect(globalThis.fetch).not.toHaveBeenCalled(); + }); + + it('falls back to GitHub when the database has no series for the run', async () => { + process.env.DATABASE_READONLY_URL = 'postgresql://readonly.example.test/db'; + mockGetGpuMetricsForRun.mockResolvedValueOnce(null); + globalThis.fetch = vi + .fn() + .mockResolvedValueOnce({ + ok: true, + json: () => + Promise.resolve({ + id: 99, + name: 'In progress', + head_branch: 'main', + head_sha: 'abc', + created_at: '2026-09-17T00:00:00Z', + html_url: 'https://github.com/TestOwner/TestRepo/actions/runs/99', + conclusion: null, + status: 'in_progress', + }), + }) + .mockResolvedValueOnce({ + ok: true, + json: () => + Promise.resolve({ + artifacts: [ + { id: 1, name: 'gpu_metrics_live', archive_download_url: 'https://example.com/dl/1' }, + ], + }), + }) + .mockResolvedValueOnce({ + ok: true, + headers: new Headers({ 'Content-Length': '1024' }), + arrayBuffer: () => Promise.resolve(new ArrayBuffer(0)), + }); + + const res = await GET(req('/api/gpu-metrics?runId=99')); + expect(res.status).toBe(200); + const body = await res.json(); + expect(body.source).toBe('github'); + expect(body.artifacts[0].name).toBe('gpu_metrics_live'); + }); + + it('falls back to GitHub when the database lookup throws', async () => { + process.env.DATABASE_READONLY_URL = 'postgresql://readonly.example.test/db'; + mockGetGpuMetricsForRun.mockRejectedValueOnce(new Error('relation does not exist')); + globalThis.fetch = vi.fn().mockResolvedValueOnce({ ok: false, status: 404 }); + + const res = await GET(req('/api/gpu-metrics?runId=99')); + expect(res.status).toBe(500); + expect(globalThis.fetch).toHaveBeenCalledTimes(1); + }); }); describe('GET /api/gpu-metrics', () => { diff --git a/packages/app/src/app/api/gpu-metrics/route.ts b/packages/app/src/app/api/gpu-metrics/route.ts index 59a607f35..e523ec2b8 100644 --- a/packages/app/src/app/api/gpu-metrics/route.ts +++ b/packages/app/src/app/api/gpu-metrics/route.ts @@ -1,10 +1,29 @@ /** - * DO NOT ADD CACHING (blob, CDN, or unstable_cache) to this route. - * It fetches live GitHub Actions artifacts which change while a run is in progress. + * PowerX explorer data for one GitHub Actions run. + * + * Reads the ingest-time telemetry digest first (migration 016: series, + * samples, per-GPU statistics, point links). Runs that have not been ingested + * yet, including runs still in progress, fall back to the live GitHub + * artifacts exactly as before. + * + * DO NOT ADD CACHING (blob, CDN, or unstable_cache) to this route. The + * fallback fetches live GitHub Actions artifacts which change while a run is + * in progress, and the database path is already a single indexed read. */ import { type NextRequest, NextResponse } from 'next/server'; -import { parseCsvData } from '@/components/gpu-power/types'; +import { getDb } from '@semianalysisai/inferencex-db/connection'; +import { + getGpuMetricsForRun, + type GpuMetricSeries, + type GpuMetricsRunPayload, +} from '@semianalysisai/inferencex-db/queries/gpu-metrics'; + +import { + parseCsvData, + type GpuMetricRow, + type GpuPowerRunInfo, +} from '@/components/gpu-power/types'; import { downloadGithubArtifact, extractZipEntries, @@ -17,7 +36,55 @@ import { const MAX_ARTIFACT_BYTES = 50 * 1024 * 1024; -async function fetchGpuMetrics(runId: string) { +export type GpuMetricsSource = 'database' | 'github'; + +export interface GpuMetricsArtifactPayload { + name: string; + data: GpuMetricRow[]; + /** Present only for database-backed artifacts. */ + series?: Omit; +} + +export interface GpuMetricsRouteResponse { + runInfo: GpuPowerRunInfo; + artifacts: GpuMetricsArtifactPayload[]; + source: GpuMetricsSource; +} + +/** Shape the stored digest like the GitHub payload so the explorer is source-agnostic. */ +export function databasePayloadToResponse(payload: GpuMetricsRunPayload): GpuMetricsRouteResponse { + const run = payload.workflowRun; + const filesPerArtifact = new Map(); + for (const series of payload.series) { + filesPerArtifact.set(series.artifactName, (filesPerArtifact.get(series.artifactName) ?? 0) + 1); + } + return { + source: 'database', + runInfo: { + id: run.githubRunId, + name: run.name, + branch: run.headBranch ?? '', + sha: run.headSha ?? '', + createdAt: run.createdAt ?? `${run.date}T00:00:00Z`, + url: + run.htmlUrl ?? + `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${run.githubRunId}`, + conclusion: run.conclusion ?? '', + status: run.status ?? '', + }, + artifacts: payload.series.map(({ data, ...series }) => ({ + // Multinode uploads carry one CSV per node; keep them distinguishable. + name: + (filesPerArtifact.get(series.artifactName) ?? 1) > 1 + ? `${series.artifactName}/${series.fileName}` + : series.artifactName, + data, + series, + })), + }; +} + +async function fetchGpuMetricsFromGithub(runId: string): Promise { const githubToken = getGithubToken(); if (!githubToken) throw new Error('GitHub token not configured'); @@ -30,7 +97,7 @@ async function fetchGpuMetrics(runId: string) { const gpuArtifacts = artifacts.filter((a) => a.name.startsWith('gpu_metrics')); if (gpuArtifacts.length === 0) throw new Error('No gpu_metrics artifacts found for this run'); - const parsedArtifacts: { name: string; data: ReturnType }[] = []; + const parsedArtifacts: GpuMetricsArtifactPayload[] = []; for (const artifact of gpuArtifacts) { const dlResp = await downloadGithubArtifact(artifact.archive_download_url, githubToken); if (!dlResp.ok) { @@ -58,11 +125,25 @@ async function fetchGpuMetrics(runId: string) { if (parsedArtifacts.length === 0) throw new Error('No Chip metrics data found in artifacts'); return { - runInfo: normalizeGithubRunInfo(run), + source: 'github', + runInfo: normalizeGithubRunInfo(run) as GpuPowerRunInfo, artifacts: parsedArtifacts, }; } +async function fetchGpuMetricsFromDatabase(runId: string): Promise { + if (!process.env.DATABASE_READONLY_URL) return null; + try { + const payload = await getGpuMetricsForRun(getDb(), Number(runId)); + return payload ? databasePayloadToResponse(payload) : null; + } catch (error) { + // A schema that predates migration 016 or a transient DB error must not + // hide the live GitHub artifacts. + console.warn(`gpu-metrics: database lookup failed for run ${runId}, using GitHub:`, error); + return null; + } +} + export async function GET(request: NextRequest) { const runId = request.nextUrl.searchParams.get('runId'); @@ -71,7 +152,8 @@ export async function GET(request: NextRequest) { } try { - const data = await fetchGpuMetrics(runId); + const data = + (await fetchGpuMetricsFromDatabase(runId)) ?? (await fetchGpuMetricsFromGithub(runId)); return NextResponse.json(data); } catch (error) { console.error('Error fetching GPU power data:', error); diff --git a/packages/app/src/app/api/v1/gpu-metrics-point/route.ts b/packages/app/src/app/api/v1/gpu-metrics-point/route.ts new file mode 100644 index 000000000..f9d04f1df --- /dev/null +++ b/packages/app/src/app/api/v1/gpu-metrics-point/route.ts @@ -0,0 +1,39 @@ +import { getDb } from '@semianalysisai/inferencex-db/connection'; +import { + getGpuMetricsForPoint, + type GpuMetricsPointPayload, +} from '@semianalysisai/inferencex-db/queries/gpu-metrics'; + +import { cachedQuery } from '@/lib/api-cache'; + +import { idQueryRoute } from '../id-routes'; + +export const dynamic = 'force-dynamic'; + +/** + * Blob-cache namespace. Stored series are immutable per (run, artifact, CSV + * hash), so the payload for a point only changes when the digest schema does; + * bump the suffix alongside any change to the row shape in + * `queries/gpu-metrics.ts`. + */ +export const CACHE_KEY_PREFIX = 'gpu-metrics-point-v1'; + +const getCachedGpuMetricsForPoint = cachedQuery( + (id: number): Promise => getGpuMetricsForPoint(getDb(), id), + CACHE_KEY_PREFIX, + { blobOnly: true }, +); + +/** + * GET /api/v1/gpu-metrics-point?id=N + * + * PowerX telemetry recorded while one benchmark point ran: every linked + * gpu_metrics series with full-resolution per-GPU samples and the ingest-time + * per-GPU statistics digest. 404 when the point has no linked series (its + * run predates migration 016 and the artifacts have expired, or the job + * uploaded no gpu_metrics artifact). + */ +export const GET = idQueryRoute({ + logLabel: 'gpu metrics point', + fetch: getCachedGpuMetricsForPoint, +}); diff --git a/packages/app/src/components/inference/agentic-point/agentic-point-detail.tsx b/packages/app/src/components/inference/agentic-point/agentic-point-detail.tsx index 369078c40..f78e3b2af 100644 --- a/packages/app/src/components/inference/agentic-point/agentic-point-detail.tsx +++ b/packages/app/src/components/inference/agentic-point/agentic-point-detail.tsx @@ -33,6 +33,7 @@ import { type StagePhase, } from './phase-slice'; import { PointSummary } from './point-summary'; +import { PowerTelemetryView } from './power-telemetry-view'; import { RequestMetricOverTime, SequenceMetricCard } from './request-metric-cards'; import { ServerLogViewer } from './server-log-viewer'; import { @@ -61,6 +62,7 @@ export const AGENTIC_POINT_DETAIL_STRINGS = { ttftOverTime: 'TTFT over time', perPoint: 'Per-point', requestTimeline: 'Request timeline', + powerX: 'PowerX', aggregatesAcrossConfigs: 'Aggregates across configs', logs: 'Logs', detailView: 'Detail view', @@ -94,6 +96,7 @@ export const AGENTIC_POINT_DETAIL_STRINGS = { ttftOverTime: 'TTFT 随时间变化', perPoint: '单点', requestTimeline: '请求时间线', + powerX: 'PowerX', aggregatesAcrossConfigs: '跨配置聚合', logs: '日志', detailView: '详情视图', @@ -149,6 +152,7 @@ export function AgenticPointDetail({ id }: Props) { () => [ { value: 'point', label: t.perPoint, testId: 'detail-view-point' }, { value: 'timeline', label: t.requestTimeline, testId: 'detail-view-timeline' }, + { value: 'power', label: t.powerX, testId: 'detail-view-power' }, { value: 'aggregates', label: t.aggregatesAcrossConfigs, testId: 'detail-view-aggregates' }, { value: 'logs', label: t.logs, testId: 'detail-view-logs' }, ], @@ -347,6 +351,8 @@ export function AgenticPointDetail({ id }: Props) { {view === 'logs' ? ( + ) : view === 'power' ? ( + ) : view === 'aggregates' ? ( aggregatesQuery.isError ? ( = { nvidia: 'nvidia-smi', amd: 'amd-smi' }; + +interface Props { + id: number; + enabled: boolean; +} + +function seriesLabel(series: GpuMetricSeries, total: number): string { + return total > 1 ? `${series.artifactName} · ${series.fileName}` : series.artifactName; +} + +/** + * PowerX tab of the per-point detail page: the full-resolution chip telemetry + * recorded while this benchmark point ran, read from the ingest-time digest + * (migration 016) rather than from GitHub artifacts. + */ +export function PowerTelemetryView({ id, enabled }: Props) { + const locale = useLocale(); + const t = STRINGS[locale]; + const query = useGpuMetricsPoint(id, enabled); + const seriesList = query.data?.series ?? []; + + const [seriesSelection, setSeriesSelection] = useState<{ id: number; seriesId: number } | null>( + null, + ); + const selectedSeries = + (seriesSelection?.id === id + ? seriesList.find((series) => series.id === seriesSelection.seriesId) + : undefined) ?? seriesList[0]; + const data: GpuMetricRow[] = useMemo(() => selectedSeries?.data ?? [], [selectedSeries]); + const availableMetrics = useMemo(() => getAvailableMetrics(data), [data]); + + const [metricSelection, setMetricSelection] = useState('power'); + const metricKey: GpuMetricKey = availableMetrics.some((m) => m.key === metricSelection) + ? metricSelection + : 'power'; + const metricConfig = ALL_METRIC_OPTIONS.find((m) => m.key === metricKey)!; + const visibleGpus = useMemo(() => new Set(data.map((row) => row.index)), [data]); + + if (!enabled) return null; + + if (query.isLoading) { + return ( +
+ {t.loading} +
+ ); + } + if (query.isError) { + return ( + + ); + } + if (!selectedSeries) { + return ( +
+ {t.missing.replace('{id}', String(id))} +
+ ); + } + + const durationS = Math.max( + 0, + (new Date(selectedSeries.endedAt).getTime() - new Date(selectedSeries.startedAt).getTime()) / + 1000, + ); + const numberLocale = locale === 'zh' ? 'zh-CN' : undefined; + + return ( +
+ +
+
+
{t.vendor}
+
+ {VENDOR_LABEL[selectedSeries.vendor] ?? selectedSeries.vendor} +
+
+
+
{t.samples}
+
+ {selectedSeries.sampleCount.toLocaleString(numberLocale)} +
+
+
+
{t.chips}
+
{selectedSeries.gpuCount}
+
+
+
{t.interval}
+
+ {selectedSeries.sampleIntervalS === null + ? '—' + : `${selectedSeries.sampleIntervalS.toFixed(2)} ${t.secondsUnit}`} +
+
+
+
{t.window}
+
+ {new Date(selectedSeries.startedAt).toLocaleTimeString(numberLocale)} ·{' '} + {Math.round(durationS).toLocaleString(numberLocale)} {t.secondsUnit} +
+
+
+
+ {seriesList.length > 1 && ( +
+ + +
+ )} +
+ + +
+
+
+ + + + {getGpuMetricLabel(metricConfig, locale)} · {t.sharedNote} + + } + /> + + + +

{t.perGpuStats}

+ +
+
+ ); +} diff --git a/packages/app/src/components/inference/agentic-point/use-detail-view.test.tsx b/packages/app/src/components/inference/agentic-point/use-detail-view.test.tsx index 4fe54b399..64874bb74 100644 --- a/packages/app/src/components/inference/agentic-point/use-detail-view.test.tsx +++ b/packages/app/src/components/inference/agentic-point/use-detail-view.test.tsx @@ -21,6 +21,7 @@ function DetailViewProbe() { createElement('button', { onClick: () => setView('point') }, 'point'), createElement('button', { onClick: () => setView('timeline') }, 'timeline'), createElement('button', { onClick: () => setView('aggregates') }, 'aggregates'), + createElement('button', { onClick: () => setView('power') }, 'power'), ); } @@ -119,4 +120,25 @@ describe('useDetailView', () => { }); expect(renderedView()).toBe('aggregates'); }); + + it('accepts the PowerX view from the URL and writes it back on selection', () => { + nativeReplace('/inference/agentic/42?view=power'); + renderProbe(); + expect(renderedView()).toBe('power'); + + click('point'); + expect(new URLSearchParams(window.location.search).has('view')).toBe(false); + click('power'); + expect(renderedView()).toBe('power'); + expect(new URLSearchParams(window.location.search).get('view')).toBe('power'); + expect(track).toHaveBeenLastCalledWith('inference_agentic_detail_view_changed', { + view: 'power', + }); + }); + + it('falls back to the per-point view for unknown view names', () => { + nativeReplace('/inference/agentic/42?view=telemetry'); + renderProbe(); + expect(renderedView()).toBe('point'); + }); }); diff --git a/packages/app/src/components/inference/agentic-point/use-detail-view.ts b/packages/app/src/components/inference/agentic-point/use-detail-view.ts index 1a97a85cc..b08b3a868 100644 --- a/packages/app/src/components/inference/agentic-point/use-detail-view.ts +++ b/packages/app/src/components/inference/agentic-point/use-detail-view.ts @@ -6,10 +6,14 @@ import { useClientSearchParams } from '@/hooks/useClientSearch'; import { track } from '@/lib/analytics'; import { replaceClientSearch } from '@/lib/client-navigation'; -export type DetailView = 'point' | 'timeline' | 'aggregates' | 'logs'; +export type DetailView = 'point' | 'timeline' | 'power' | 'aggregates' | 'logs'; const isDetailView = (value: string | null): value is DetailView => - value === 'point' || value === 'timeline' || value === 'aggregates' || value === 'logs'; + value === 'point' || + value === 'timeline' || + value === 'power' || + value === 'aggregates' || + value === 'logs'; /** URL-persisted detail view (`?view=`; per-point is the unadorned default). */ export function useDetailView(): [DetailView, (nextView: DetailView) => void] { diff --git a/packages/app/src/hooks/api/use-gpu-metrics-point.ts b/packages/app/src/hooks/api/use-gpu-metrics-point.ts new file mode 100644 index 000000000..12c023331 --- /dev/null +++ b/packages/app/src/hooks/api/use-gpu-metrics-point.ts @@ -0,0 +1,18 @@ +import type { + GpuMetricsPointPayload, + GpuMetricSeries, + GpuMetricStatRow, +} from '@semianalysisai/inferencex-db/queries/gpu-metrics'; + +import { useByIdQuery } from './benchmark-id-query'; + +export type { GpuMetricsPointPayload, GpuMetricSeries, GpuMetricStatRow }; + +/** + * Lazy-fetch the PowerX telemetry linked to one benchmark point. Enabled only + * while the PowerX detail view is open: a series is one 1 s sample per GPU for + * the whole job (hundreds of KB), so it is not paid for on every page load. + */ +export function useGpuMetricsPoint(id: number | null, enabled = false) { + return useByIdQuery('gpu-metrics-point', id, enabled && Boolean(id)); +} diff --git a/packages/constants/src/tables.ts b/packages/constants/src/tables.ts index 684f310bc..d1bcd6a0b 100644 --- a/packages/constants/src/tables.ts +++ b/packages/constants/src/tables.ts @@ -5,6 +5,10 @@ export const TABLE_NAMES = { agenticTraceReplay: 'agentic_trace_replay', benchmarkResults: 'benchmark_results', serverLogs: 'server_logs', + gpuMetricSeries: 'gpu_metric_series', + gpuMetricSamples: 'gpu_metric_samples', + gpuMetricGpuStats: 'gpu_metric_gpu_stats', + benchmarkResultGpuMetrics: 'benchmark_result_gpu_metrics', runStats: 'run_stats', evalResults: 'eval_results', evalSamples: 'eval_samples', diff --git a/packages/db/migrations/016_gpu_metrics.sql b/packages/db/migrations/016_gpu_metrics.sql new file mode 100644 index 000000000..8c3f52ac5 --- /dev/null +++ b/packages/db/migrations/016_gpu_metrics.sql @@ -0,0 +1,99 @@ +-- PowerX telemetry digest. +-- +-- The producer samples nvidia-smi / amd-smi once per second for the lifetime +-- of every benchmark job and uploads the CSV as a `gpu_metrics_` +-- artifact next to `bmk_`. Until now the app downloaded and parsed +-- those artifacts from GitHub on every page view, and lost them entirely once +-- GitHub's 90-day artifact retention expired. These tables move that work to +-- ingest time: raw samples are kept at full resolution, per-GPU summary +-- statistics are digested once, and each benchmark point is linked to the +-- series that was recorded while it ran. + +create table gpu_metric_series ( + id bigserial primary key, + workflow_run_id bigint not null references workflow_runs(id) on delete cascade, + -- Full GitHub artifact name, including the runner-pool/attempt suffix. + artifact_name text not null, + -- Artifact name with the gpu_metrics_ prefix removed; pairs with bmk_. + config_key text not null, + -- CSV path relative to the extracted artifact root (multinode uploads can + -- carry one CSV per serving node). + file_name text not null, + vendor text not null, + csv_sha256 text not null, + sample_interval_s real, + sample_count integer not null, + gpu_count smallint not null, + started_at timestamptz not null, + ended_at timestamptz not null, + -- Parsed sidecars: gpu_metrics_context.json, gpu_metrics_identity.*, + -- gpu_metrics_energy_{start,end}.csv. Kept verbatim for provenance. + sidecars jsonb not null default '{}'::jsonb, + ingested_at timestamptz not null default now(), + + constraint gpu_metric_series_vendor_known check (vendor in ('nvidia', 'amd')), + constraint gpu_metric_series_artifact_nonempty check (artifact_name <> ''), + constraint gpu_metric_series_file_nonempty check (file_name <> ''), + constraint gpu_metric_series_sample_count_non_neg check (sample_count >= 0), + constraint gpu_metric_series_gpu_count_non_neg check (gpu_count >= 0), + constraint gpu_metric_series_window_ordered check (ended_at >= started_at), + constraint gpu_metric_series_unique unique (workflow_run_id, artifact_name, file_name) +); + +create index gpu_metric_series_run_idx on gpu_metric_series (workflow_run_id); + +-- One row per (GPU, sample). Vendor-specific columns stay null when the +-- collector does not report them. `real` keeps the row narrow; the source +-- telemetry has at most three significant decimals. +create table gpu_metric_samples ( + series_id bigint not null references gpu_metric_series(id) on delete cascade, + gpu_index smallint not null, + sampled_at timestamptz not null, + power_w real, + temperature_c real, + sm_clock_mhz real, + mem_clock_mhz real, + gpu_util_pct real, + mem_util_pct real, + edge_temp_c real, + mem_temp_c real, + gfx_voltage_mv real, + soc_voltage_mv real, + mem_voltage_mv real, + fclk_mhz real, + socclk_mhz real, + mm_activity_pct real, + + primary key (series_id, gpu_index, sampled_at) +); + +-- Ingest-time digest so readers never rescan samples for summary cards. +create table gpu_metric_gpu_stats ( + series_id bigint not null references gpu_metric_series(id) on delete cascade, + gpu_index smallint not null, + metric text not null, + sample_count integer not null, + min_value real not null, + max_value real not null, + mean_value real not null, + median_value real not null, + p95_value real not null, + p99_value real not null, + stddev_value real not null, + + constraint gpu_metric_gpu_stats_metric_nonempty check (metric <> ''), + constraint gpu_metric_gpu_stats_sample_count_positive check (sample_count > 0), + primary key (series_id, gpu_index, metric) +); + +-- Benchmark point ↔ telemetry series. A point can reference several series +-- when a multinode artifact ships one CSV per node. +create table benchmark_result_gpu_metrics ( + benchmark_result_id bigint not null references benchmark_results(id) on delete cascade, + series_id bigint not null references gpu_metric_series(id) on delete cascade, + + primary key (benchmark_result_id, series_id) +); + +create index benchmark_result_gpu_metrics_series_idx + on benchmark_result_gpu_metrics (series_id); diff --git a/packages/db/package.json b/packages/db/package.json index c6c05c57b..ade8e49b4 100644 --- a/packages/db/package.json +++ b/packages/db/package.json @@ -27,6 +27,7 @@ "db:backfill-atom-kv-capacity": "bun --env-file=../../.env src/backfill-atom-kv-capacity.ts", "db:backfill-chart-series": "bun --env-file=../../.env src/backfill-chart-series.ts", "db:backfill-dataset-stats": "bun --env-file=../../.env src/backfill-dataset-stats.ts", + "db:backfill-gpu-metrics": "bun --env-file=../../.env src/backfill-gpu-metrics.ts", "db:backfill-full-response-interactivity": "bun --env-file=../../.env src/backfill-full-response-interactivity.ts", "db:backfill-request-timeline": "bun --env-file=../../.env src/backfill-request-timeline.ts", "db:backfill-runtime-metadata": "bun --env-file=../../.env src/backfill-runtime-metadata.ts", diff --git a/packages/db/src/backfill-gpu-metrics.ts b/packages/db/src/backfill-gpu-metrics.ts new file mode 100644 index 000000000..5cbfa640d --- /dev/null +++ b/packages/db/src/backfill-gpu-metrics.ts @@ -0,0 +1,303 @@ +/** + * Backfill PowerX telemetry (`gpu_metrics_` artifacts) into the + * migration-016 tables for runs that were ingested before the CI path + * digested them. + * + * GitHub keeps run artifacts for 90 days and the GCS backup only mirrors + * bmk_/server_logs_ uploads, so the reachable history is bounded by GitHub + * retention. Each gpu_metrics artifact is paired with its exact `bmk_` + * (or `bmk_agentic_`) sibling, the raw rows are mapped through the + * production mapper, and the series is linked to those persisted points. + * + * Usage: + * bun run --cwd packages/db db:backfill-gpu-metrics --run 34557177019 --yes + * bun run --cwd packages/db db:backfill-gpu-metrics --all --yes + * bun run --cwd packages/db db:backfill-gpu-metrics --all --since 2026-08-01 --dry-run + * bun run --cwd packages/db db:backfill-gpu-metrics --all --force --limit 20 --yes + * + * Runs that already have at least one stored series are skipped unless + * --force is passed. --parallel N (default 4) bounds concurrent artifact + * downloads within one run. + */ + +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { hasNoSslFlag } from './cli-utils.js'; +import { AsyncSemaphore } from './etl/async-semaphore.js'; +import { createAdminSql } from './etl/db-utils.js'; +import { ingestGpuMetricsArtifact } from './etl/gpu-metrics-ingest.js'; +import { retryArtifactOperation } from './lib/artifact-retry.js'; +import { confirmProceed, parseLimitForceFlags, runBackfillMain } from './lib/backfill-runner.js'; +import { findBenchmarkResultIds, readMappedBenchmarkRows } from './lib/benchmark-result-lookup.js'; +import { downloadArtifact, listRunArtifacts } from './lib/github-artifacts.js'; +import { + pairGpuMetricsArtifacts, + type GpuMetricsArtifactPair, +} from './lib/gpu-metrics-backfill.js'; +import { repositoryFromRunUrl } from './lib/runtime-metadata-artifacts.js'; + +const DEFAULT_REPO = 'SemiAnalysisAI/InferenceX'; +const GITHUB_RETENTION_DAYS = 90; +const sql = createAdminSql({ noSsl: hasNoSslFlag(), max: 4, onnotice: () => {} }); + +interface CandidateRun { + id: number; + github_run_id: number; + run_attempt: number; + html_url: string | null; + date: string; + series_count: number; +} + +interface BackfillFlags { + all: boolean; + dryRun: boolean; + run: number | null; + fromRun: number | null; + since: string | null; + parallel: number; +} + +function positiveIntFlag(flag: string): number | null { + const index = process.argv.indexOf(flag); + if (index === -1) return null; + const raw = process.argv[index + 1]; + if (!raw || !/^\d+$/u.test(raw) || Number(raw) <= 0) { + throw new Error(`${flag} requires a positive integer`); + } + return Number(raw); +} + +function parseFlags(): BackfillFlags { + const sinceIndex = process.argv.indexOf('--since'); + const since = sinceIndex === -1 ? null : (process.argv[sinceIndex + 1] ?? null); + if (sinceIndex !== -1 && (!since || !/^\d{4}-\d{2}-\d{2}$/u.test(since))) { + throw new Error('--since requires a YYYY-MM-DD date'); + } + return { + all: process.argv.includes('--all'), + dryRun: process.argv.includes('--dry-run'), + run: positiveIntFlag('--run'), + fromRun: positiveIntFlag('--from-run'), + since, + parallel: positiveIntFlag('--parallel') ?? 4, + }; +} + +function isWithinGithubRetention(date: string): boolean { + const ageMs = Date.now() - new Date(date).getTime(); + return ageMs <= GITHUB_RETENTION_DAYS * 24 * 60 * 60 * 1000; +} + +async function loadCandidateRuns( + flags: BackfillFlags, + limit: number | null, + force: boolean, +): Promise { + const cutoff = new Date(Date.now() - GITHUB_RETENTION_DAYS * 24 * 60 * 60 * 1000) + .toISOString() + .slice(0, 10); + const since = flags.since ?? cutoff; + const rows = await sql` + select wr.id, wr.github_run_id, wr.run_attempt, wr.html_url, wr.date::text as date, + (select count(*)::int from gpu_metric_series s where s.workflow_run_id = wr.id) as series_count + from latest_workflow_runs wr + where exists (select 1 from benchmark_results br where br.workflow_run_id = wr.id) + and (${flags.run}::bigint is null or wr.github_run_id = ${flags.run}) + and (${flags.fromRun}::bigint is null or wr.github_run_id >= ${flags.fromRun}) + and (${flags.run}::bigint is not null or wr.date >= ${since}::date) + order by wr.date desc, wr.github_run_id desc + `; + const candidates = rows + .map((row) => ({ + ...row, + id: Number(row.id), + github_run_id: Number(row.github_run_id), + series_count: Number(row.series_count), + })) + .filter((row) => force || flags.run !== null || row.series_count === 0); + return limit === null ? candidates : candidates.slice(0, limit); +} + +type PairOutcome = + | { kind: 'ingested'; seriesCount: number; samplesInserted: number; pointsLinked: number } + | { kind: 'unmatched' } + | { kind: 'empty' } + | { kind: 'failed' }; + +/** Download one gpu_metrics/bmk pair, resolve its points, and persist the series. */ +async function processPair( + run: CandidateRun, + pair: GpuMetricsArtifactPair, + tempDir: string, +): Promise { + let benchmarkDir: string | null = null; + let gpuMetricsDir: string | null = null; + try { + benchmarkDir = await retryArtifactOperation(`downloading ${pair.benchmarks.name}`, () => + downloadArtifact(pair.benchmarks, tempDir), + ); + const mappedRows = readMappedBenchmarkRows(benchmarkDir); + const resultIds = await findBenchmarkResultIds(sql, run, mappedRows); + if (resultIds.length === 0) { + console.warn(` [WARN] ${pair.gpuMetrics.name}: no matching benchmark rows`); + return { kind: 'unmatched' }; + } + gpuMetricsDir = await retryArtifactOperation(`downloading ${pair.gpuMetrics.name}`, () => + downloadArtifact(pair.gpuMetrics, tempDir), + ); + const ingested = await ingestGpuMetricsArtifact(sql, { + workflowRunId: run.id, + artifact: { artifactName: pair.gpuMetrics.name, artifactDir: gpuMetricsDir }, + benchmarkResultIds: resultIds, + }); + if (ingested.seriesIds.length === 0) { + console.warn(` [WARN] ${pair.gpuMetrics.name}: no parseable gpu_metrics CSV`); + return { kind: 'empty' }; + } + return { + kind: 'ingested', + seriesCount: ingested.seriesIds.length, + samplesInserted: ingested.samplesInserted, + pointsLinked: resultIds.length, + }; + } catch (error) { + console.error(` ✗ run ${run.github_run_id} artifact ${pair.gpuMetrics.name}:`, error); + return { kind: 'failed' }; + } finally { + if (benchmarkDir) fs.rmSync(benchmarkDir, { recursive: true, force: true }); + if (gpuMetricsDir) fs.rmSync(gpuMetricsDir, { recursive: true, force: true }); + } +} + +async function main(): Promise { + const flags = parseFlags(); + const { limit, force } = parseLimitForceFlags(); + if (!flags.all && flags.run === null) { + throw new Error('Pass --run or --all'); + } + + console.log('=== backfill-gpu-metrics ==='); + const runs = await loadCandidateRuns(flags, limit, force); + const staleRuns = runs.filter((run) => !isWithinGithubRetention(run.date)).length; + const staleNote = + staleRuns > 0 + ? ` (${staleRuns} older than GitHub's ${GITHUB_RETENTION_DAYS}-day retention)` + : ''; + console.log(` ${runs.length} candidate run(s)${staleNote}`); + if (runs.length === 0) { + console.log('\n Nothing to do.'); + return; + } + + if (flags.dryRun) { + let pairedRuns = 0; + let pairs = 0; + for (const run of runs) { + const repository = repositoryFromRunUrl(run.html_url) ?? DEFAULT_REPO; + const artifacts = await retryArtifactOperation( + `listing GitHub artifacts for run ${run.github_run_id}`, + () => listRunArtifacts(repository, String(run.github_run_id)), + ); + const runPairs = pairGpuMetricsArtifacts(artifacts); + if (runPairs.length > 0) pairedRuns++; + pairs += runPairs.length; + console.log(` run ${run.github_run_id} (${run.date}): ${runPairs.length} pair(s)`); + } + console.log( + `\n=== dry run: ${runs.length} run(s), ${pairedRuns} with pairs, ${pairs} pair(s) ===`, + ); + return; + } + + if (!(await confirmProceed(`${runs.length} workflow run(s) will be checked for gpu_metrics.`))) { + return; + } + + let artifactsProcessed = 0; + let seriesStored = 0; + let samplesStored = 0; + let pointsLinked = 0; + let unmatchedArtifacts = 0; + let emptyArtifacts = 0; + let artifactFailures = 0; + let runFailures = 0; + let missingRuns = 0; + + for (const [runIndex, run] of runs.entries()) { + const runId = run.github_run_id; + const repository = repositoryFromRunUrl(run.html_url) ?? DEFAULT_REPO; + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `gpu-metrics-backfill-${runId}-`)); + const runStart = Date.now(); + try { + const artifacts = await retryArtifactOperation( + `listing GitHub artifacts for run ${runId}`, + () => listRunArtifacts(repository, String(runId)), + ); + const pairs = pairGpuMetricsArtifacts(artifacts); + if (pairs.length === 0) { + missingRuns++; + console.log( + ` [${runIndex + 1}/${runs.length}] run ${runId} attempt ${run.run_attempt}: no retained gpu_metrics pairs`, + ); + continue; + } + + const limiter = new AsyncSemaphore(flags.parallel); + const outcomes = await Promise.all( + pairs.map((pair) => limiter.run(() => processPair(run, pair, tempDir))), + ); + let runSeries = 0; + let runSamples = 0; + for (const outcome of outcomes) { + switch (outcome.kind) { + case 'ingested': { + artifactsProcessed++; + runSeries += outcome.seriesCount; + runSamples += outcome.samplesInserted; + pointsLinked += outcome.pointsLinked; + break; + } + case 'unmatched': { + unmatchedArtifacts++; + break; + } + case 'empty': { + emptyArtifacts++; + break; + } + case 'failed': { + artifactFailures++; + break; + } + } + } + seriesStored += runSeries; + samplesStored += runSamples; + console.log( + ` [${runIndex + 1}/${runs.length}] run ${runId} attempt ${run.run_attempt} (${run.date}): ` + + `${pairs.length} pair(s), ${runSeries} series, +${runSamples} samples ` + + `(${((Date.now() - runStart) / 1000).toFixed(1)}s)`, + ); + } catch (error) { + runFailures++; + console.error(` ✗ run ${runId}:`, error); + } finally { + fs.rmSync(tempDir, { recursive: true, force: true }); + } + } + + console.log( + `\n=== backfill complete: ${artifactsProcessed} artifact(s), ${seriesStored} series, ` + + `${samplesStored} sample(s), ${pointsLinked} point link(s), ` + + `${unmatchedArtifacts} unmatched artifact(s), ${emptyArtifacts} empty artifact(s), ` + + `${missingRuns} run(s) without pairs, ${artifactFailures} failed artifact(s), ` + + `${runFailures} failed run(s) ===`, + ); + console.log(' Invalidate API cache after the backfill: bun run admin:cache:invalidate'); + if (artifactFailures > 0 || runFailures > 0) process.exitCode = 1; +} + +runBackfillMain('backfill-gpu-metrics', sql, main); diff --git a/packages/db/src/backfill-server-log-files.ts b/packages/db/src/backfill-server-log-files.ts index 8047e0db5..3208194a2 100644 --- a/packages/db/src/backfill-server-log-files.ts +++ b/packages/db/src/backfill-server-log-files.ts @@ -19,11 +19,9 @@ import os from 'node:os'; import path from 'node:path'; import { hasNoSslFlag } from './cli-utils.js'; -import { mapBenchmarkRow, type BenchmarkParams } from './etl/benchmark-mapper.js'; import { insertServerLogFilePaths } from './etl/benchmark-ingest.js'; import { createAdminSql } from './etl/db-utils.js'; import { listServerLogFilePaths, serverLogArtifactRoot } from './etl/server-log-artifacts.js'; -import { createSkipTracker } from './etl/skip-tracker.js'; import { downloadArtifact, listRunArtifacts } from './lib/github-artifacts.js'; import { downloadGcsArtifact, @@ -32,11 +30,9 @@ import { } from './lib/gcs-artifacts.js'; import { confirmProceed, parseLimitForceFlags, runBackfillMain } from './lib/backfill-runner.js'; import { retryArtifactOperation } from './lib/artifact-retry.js'; +import { findBenchmarkResultIds, readMappedBenchmarkRows } from './lib/benchmark-result-lookup.js'; import { repositoryFromRunUrl } from './lib/runtime-metadata-artifacts.js'; -import { - pairServerLogArtifacts, - resolveServerLogResultCandidates, -} from './lib/server-log-backfill.js'; +import { pairServerLogArtifacts } from './lib/server-log-backfill.js'; const DEFAULT_REPO = 'SemiAnalysisAI/InferenceX'; const RETENTION_DAYS = 90; @@ -94,86 +90,6 @@ function parseBackfillFlags(): BackfillFlags { }; } -function findJsonFiles(root: string): string[] { - const files: string[] = []; - for (const entry of fs.readdirSync(root, { withFileTypes: true })) { - const pathname = path.join(root, entry.name); - if (entry.isDirectory()) files.push(...findJsonFiles(pathname)); - else if (entry.isFile() && entry.name.endsWith('.json')) files.push(pathname); - } - return files.toSorted(); -} - -function readMappedRows(root: string): BenchmarkParams[] { - const tracker = createSkipTracker(); - const rows: BenchmarkParams[] = []; - for (const file of findJsonFiles(root)) { - const parsed = JSON.parse(fs.readFileSync(file, 'utf8')) as unknown; - const rawRows = Array.isArray(parsed) ? parsed : [parsed]; - for (const raw of rawRows) { - if (!raw || typeof raw !== 'object' || Array.isArray(raw)) continue; - const mapped = mapBenchmarkRow(raw as Record, tracker); - if (mapped) rows.push(mapped); - } - } - return rows; -} - -async function findBenchmarkResultIds( - run: CandidateRun, - rows: readonly BenchmarkParams[], -): Promise { - const ids = new Set(); - for (const row of rows) { - const c = row.config; - const candidates = await sql<{ id: number; offload_mode: string }[]>` - select br.id, br.offload_mode - from benchmark_results br - join workflow_runs wr on wr.id = br.workflow_run_id - join configs cfg on cfg.id = br.config_id - where wr.github_run_id = ${run.github_run_id} - and wr.run_attempt = ${run.run_attempt} - and cfg.hardware = ${c.hardware} - and cfg.framework = ${c.framework} - and cfg.model = ${c.model} - and cfg.precision = ${c.precision} - and cfg.spec_method = ${c.specMethod} - and cfg.disagg = ${c.disagg} - and cfg.is_multinode = ${c.isMultinode} - and cfg.prefill_tp = ${c.prefillTp} - and cfg.prefill_ep = ${c.prefillEp} - and cfg.prefill_dp_attention = ${c.prefillDpAttn} - and cfg.prefill_num_workers = ${c.prefillNumWorkers} - and cfg.decode_tp = ${c.decodeTp} - and cfg.decode_ep = ${c.decodeEp} - and cfg.decode_dp_attention = ${c.decodeDpAttn} - and cfg.decode_num_workers = ${c.decodeNumWorkers} - and cfg.num_prefill_gpu = ${c.numPrefillGpu} - and cfg.num_decode_gpu = ${c.numDecodeGpu} - and br.benchmark_type = ${row.benchmarkType} - and br.isl is not distinct from ${row.isl} - and br.osl is not distinct from ${row.osl} - and br.conc = ${row.conc} - and br.recipe_fingerprint is not distinct from ${row.recipeFingerprint} - `; - const resolution = resolveServerLogResultCandidates( - candidates.map((candidate) => ({ - id: Number(candidate.id), - offloadMode: candidate.offload_mode, - })), - row.offloadMode, - ); - if (resolution.usedUniqueFallback) { - console.warn( - ` [WARN] benchmark result ${resolution.ids[0]} uses a historical offload label; ` + - `matched uniquely without offload_mode`, - ); - } - for (const id of resolution.ids) ids.add(id); - } - return [...ids]; -} - async function resultLogsAreComplete(resultIds: readonly number[]): Promise { if (resultIds.length === 0) return false; const [row] = await sql<{ complete: boolean }[]>` @@ -311,8 +227,13 @@ async function main(): Promise { : await retryArtifactOperation(`downloading ${pair.benchmarks.name}`, () => downloadArtifact(pair.benchmarks, tempDir), ); - const mappedRows = readMappedRows(benchmarkDir); - const resultIds = await findBenchmarkResultIds(run, mappedRows); + const mappedRows = readMappedBenchmarkRows(benchmarkDir); + const resultIds = await findBenchmarkResultIds(sql, run, mappedRows, (id) => + console.warn( + ` [WARN] benchmark result ${id} uses a historical offload label; ` + + `matched uniquely without offload_mode`, + ), + ); if (resultIds.length === 0) { unmatchedArtifacts++; console.warn(` [WARN] ${pair.serverLogs.name}: no matching benchmark rows`); diff --git a/packages/db/src/etl/gpu-metrics-artifacts.test.ts b/packages/db/src/etl/gpu-metrics-artifacts.test.ts new file mode 100644 index 000000000..f5efe03d4 --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-artifacts.test.ts @@ -0,0 +1,98 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { afterEach, describe, expect, it } from 'vitest'; + +import { + contextUtcOffsetMinutes, + discoverGpuMetricsArtifacts, + gpuMetricsArtifactSuffix, + listGpuMetricsCsvFiles, + parseEnergyCsv, + readGpuMetricsSidecars, +} from './gpu-metrics-artifacts.js'; + +const roots: string[] = []; + +afterEach(() => { + for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); +}); + +function tempRoot(): string { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gpu-metrics-artifacts-')); + roots.push(root); + return root; +} + +describe('gpu_metrics artifact discovery', () => { + it('pairs by the bare suffix and ignores eval-only uploads', () => { + expect(gpuMetricsArtifactSuffix('gpu_metrics_dsr1_8k1k_conc1_b200-x_0')).toBe( + 'dsr1_8k1k_conc1_b200-x_0', + ); + expect(gpuMetricsArtifactSuffix('eval_gpu_metrics_dsr1_8k1k_conc1_b200-x_0')).toBeNull(); + expect(gpuMetricsArtifactSuffix('bmk_dsr1')).toBeNull(); + }); + + it('lists telemetry CSVs recursively but skips identity and energy sidecars', () => { + const root = tempRoot(); + fs.mkdirSync(path.join(root, 'results')); + fs.writeFileSync(path.join(root, 'gpu_metrics.csv'), 'timestamp\n'); + fs.writeFileSync(path.join(root, 'gpu_metrics_identity.csv'), 'index\n'); + fs.writeFileSync(path.join(root, 'results', 'gpu_metrics_energy_start.csv'), 'gpu\n'); + fs.writeFileSync(path.join(root, 'results', 'gpu_metrics_rank1.csv'), 'timestamp\n'); + fs.writeFileSync(path.join(root, 'results', 'server.log'), ''); + + expect(listGpuMetricsCsvFiles(root).map((file) => file.fileName)).toEqual([ + 'gpu_metrics.csv', + 'results/gpu_metrics_rank1.csv', + ]); + }); + + it('indexes extracted artifact directories by suffix', () => { + const root = tempRoot(); + fs.mkdirSync(path.join(root, 'gpu_metrics_cfg-a_runner_0')); + fs.mkdirSync(path.join(root, 'bmk_cfg-a_runner_0')); + fs.writeFileSync(path.join(root, 'gpu_metrics_not-a-dir_0'), ''); + const discovered = discoverGpuMetricsArtifacts(root); + expect([...discovered.keys()]).toEqual(['cfg-a_runner_0']); + expect(discovered.get('cfg-a_runner_0')?.artifactName).toBe('gpu_metrics_cfg-a_runner_0'); + }); +}); + +describe('sidecars', () => { + it('reads context, identity CSV, and energy counters next to the CSV', () => { + const root = tempRoot(); + const csv = path.join(root, 'gpu_metrics.csv'); + fs.writeFileSync(csv, 'timestamp\n'); + fs.writeFileSync(path.join(root, 'gpu_metrics_context.json'), '{"timestamp_timezone":"UTC"}'); + fs.writeFileSync( + path.join(root, 'gpu_metrics_identity.csv'), + 'index, uuid, name\n0, GPU-abc, NVIDIA B200\n', + ); + fs.writeFileSync( + path.join(root, 'gpu_metrics_energy_start.csv'), + 'gpu,total_energy_consumption\n0,162815801.936\n', + ); + const sidecars = readGpuMetricsSidecars(csv); + expect(sidecars.context).toEqual({ timestamp_timezone: 'UTC' }); + expect(sidecars.identity).toEqual([{ index: '0', uuid: 'GPU-abc', name: 'NVIDIA B200' }]); + expect(sidecars.energyStart).toEqual({ '0': 162815801.936 }); + expect(sidecars.energyEnd).toBeNull(); + }); + + it('parses energy CSVs and rejects header-only files', () => { + expect(parseEnergyCsv('gpu,total_energy_consumption\n0,1.5\n1,2.5\n')).toEqual({ + '0': 1.5, + '1': 2.5, + }); + expect(parseEnergyCsv('gpu,total_energy_consumption\n')).toBeNull(); + }); + + it('derives a fixed UTC offset only from ±HH:MM zones', () => { + expect(contextUtcOffsetMinutes({ timestamp_timezone: 'UTC' })).toBe(0); + expect(contextUtcOffsetMinutes({ timestamp_timezone: '-05:00' })).toBe(-300); + expect(contextUtcOffsetMinutes({ timestamp_timezone: '+0530' })).toBe(330); + expect(contextUtcOffsetMinutes(null)).toBe(0); + }); +}); diff --git a/packages/db/src/etl/gpu-metrics-artifacts.ts b/packages/db/src/etl/gpu-metrics-artifacts.ts new file mode 100644 index 000000000..885c6f0af --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-artifacts.ts @@ -0,0 +1,173 @@ +/** + * Filesystem discovery for `gpu_metrics_` artifacts. + * + * Layout uploaded by the producer (`benchmark-tmpl.yml` "Upload GPU metrics"): + * gpu_metrics.csv fixed-sequence jobs + * results/gpu_metrics.csv AgentX jobs (per-concurrency) + * results/gpu_metrics*.csv multinode staging, one CSV per node + * gpu_metrics*_context.json collector context (timestamp zone …) + * gpu_metrics_identity.{json,csv} SKU / UUID / driver per GPU + * gpu_metrics_energy_{start,end}.csv amd-smi energy counters + * + * `eval_gpu_metrics_` (eval-only jobs) is deliberately ignored: those + * jobs produce no benchmark point to attach the telemetry to. + */ + +import fs from 'node:fs'; +import path from 'node:path'; + +export const GPU_METRICS_ARTIFACT_PREFIX = 'gpu_metrics_'; + +export interface GpuMetricsArtifact { + artifactName: string; + artifactDir: string; +} + +export interface GpuMetricsCsvFile { + /** POSIX-style path relative to the artifact root. */ + fileName: string; + path: string; +} + +export interface GpuMetricsSidecars { + context: Record | null; + identity: unknown | null; + energyStart: Record | null; + energyEnd: Record | null; +} + +/** Return the shared suffix that pairs a gpu_metrics artifact with bmk[_agentic]_. */ +export function gpuMetricsArtifactSuffix(artifactName: string): string | null { + return artifactName.startsWith(GPU_METRICS_ARTIFACT_PREFIX) + ? artifactName.slice(GPU_METRICS_ARTIFACT_PREFIX.length) + : null; +} + +function isGpuMetricsCsvName(fileName: string): boolean { + const lower = fileName.toLowerCase(); + if (!lower.startsWith('gpu_metrics') || !lower.endsWith('.csv')) return false; + // Sidecars share the prefix but are not time series. + return !lower.includes('_identity') && !lower.includes('_energy_'); +} + +/** Recursively list every telemetry CSV under an extracted artifact root. */ +export function listGpuMetricsCsvFiles(root: string): GpuMetricsCsvFile[] { + if (!fs.existsSync(root) || !fs.statSync(root).isDirectory()) return []; + const files: GpuMetricsCsvFile[] = []; + const visit = (directory: string): void => { + for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { + const pathname = path.join(directory, entry.name); + if (entry.isDirectory()) visit(pathname); + else if (entry.isFile() && isGpuMetricsCsvName(entry.name)) { + files.push({ + fileName: path.relative(root, pathname).split(path.sep).join('/'), + path: pathname, + }); + } + } + }; + visit(root); + return files.toSorted((a, b) => a.fileName.localeCompare(b.fileName)); +} + +function readJsonIfPresent(pathname: string): unknown | null { + if (!fs.existsSync(pathname)) return null; + try { + return JSON.parse(fs.readFileSync(pathname, 'utf8')) as unknown; + } catch { + return null; + } +} + +/** `gpu,total_energy_consumption` two-column CSV → { "": joules }. */ +export function parseEnergyCsv(text: string): Record | null { + const lines = text + .split('\n') + .map((line) => line.trim()) + .filter((line) => line.length > 0); + if (lines.length <= 1) return null; + const out: Record = {}; + for (const line of lines.slice(1)) { + const [gpu, value] = line.split(','); + const parsed = Number.parseFloat(value ?? ''); + if (gpu !== undefined && gpu !== '' && Number.isFinite(parsed)) out[gpu.trim()] = parsed; + } + return Object.keys(out).length > 0 ? out : null; +} + +/** Identity CSV (nvidia-smi) → array of column objects; JSON identity passes through. */ +function readIdentity(csvDir: string): unknown | null { + const json = readJsonIfPresent(path.join(csvDir, 'gpu_metrics_identity.json')); + if (json !== null) return json; + const csvPath = path.join(csvDir, 'gpu_metrics_identity.csv'); + if (!fs.existsSync(csvPath)) return null; + const lines = fs + .readFileSync(csvPath, 'utf8') + .split('\n') + .map((line) => line.trim()) + .filter((line) => line.length > 0); + if (lines.length <= 1) return null; + const header = lines[0]!.split(',').map((h) => h.trim()); + return lines.slice(1).map((line) => { + const cells = line.split(',').map((c) => c.trim()); + return Object.fromEntries(header.map((key, i) => [key, cells[i] ?? ''])); + }); +} + +/** + * Sidecars live next to the CSV they describe. The context file is either + * `gpu_metrics_context.json` or `_gpu_metrics_context.json`. + */ +export function readGpuMetricsSidecars(csvPath: string): GpuMetricsSidecars { + const dir = path.dirname(csvPath); + let context: Record | null = null; + for (const entry of fs.readdirSync(dir)) { + if (!entry.toLowerCase().endsWith('_context.json') || !entry.includes('gpu_metrics')) continue; + const parsed = readJsonIfPresent(path.join(dir, entry)); + if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) { + context = parsed as Record; + break; + } + } + const energyStartPath = path.join(dir, 'gpu_metrics_energy_start.csv'); + const energyEndPath = path.join(dir, 'gpu_metrics_energy_end.csv'); + return { + context, + identity: readIdentity(dir), + energyStart: fs.existsSync(energyStartPath) + ? parseEnergyCsv(fs.readFileSync(energyStartPath, 'utf8')) + : null, + energyEnd: fs.existsSync(energyEndPath) + ? parseEnergyCsv(fs.readFileSync(energyEndPath, 'utf8')) + : null, + }; +} + +/** + * Collector clock offset in minutes east of UTC, from the context sidecar. + * The producer writes `{"timestamp_timezone":"UTC"}`; anything else that is + * not a fixed `±HH:MM` offset is treated as UTC because nvidia-smi timestamps + * carry no zone of their own. + */ +export function contextUtcOffsetMinutes(context: Record | null): number { + const zone = context?.timestamp_timezone; + if (typeof zone !== 'string') return 0; + const match = /^(?[+-])(?\d{2}):?(?\d{2})$/u.exec(zone.trim()); + if (!match?.groups) return 0; + const minutes = Number(match.groups.h) * 60 + Number(match.groups.m); + return match.groups.sign === '-' ? -minutes : minutes; +} + +/** Index every extracted gpu_metrics artifact by its shared suffix. */ +export function discoverGpuMetricsArtifacts(artifactsDir: string): Map { + const discovered = new Map(); + if (!fs.existsSync(artifactsDir)) return discovered; + for (const artifactName of fs.readdirSync(artifactsDir)) { + const suffix = gpuMetricsArtifactSuffix(artifactName); + if (!suffix) continue; + const artifactDir = path.join(artifactsDir, artifactName); + if (!fs.statSync(artifactDir).isDirectory()) continue; + discovered.set(suffix, { artifactName, artifactDir }); + } + return discovered; +} diff --git a/packages/db/src/etl/gpu-metrics-csv.test.ts b/packages/db/src/etl/gpu-metrics-csv.test.ts new file mode 100644 index 000000000..2a2e2d0e0 --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-csv.test.ts @@ -0,0 +1,143 @@ +import { describe, expect, it } from 'vitest'; + +import { + computeGpuMetricStats, + parseAmdTimestamp, + parseGpuMetricsCsv, + parseMetricCell, + parseNvidiaTimestamp, + summarizeGpuMetricSamples, +} from './gpu-metrics-csv.js'; + +const NVIDIA_CSV = [ + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]', + '2026/09/11 04:19:41.982, 0, 187.80 W, 33, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:41.986, 1, 190.96 W, 39, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:42.990, 0, 912.10 W, 61, 1965 MHz, 3996 MHz, 98 %, 74 %', + '2026/09/11 04:19:42.994, 1, [N/A], 62, 1965 MHz, 3996 MHz, 97 %, 73 %', + '2026/09/11 04:19:43.990, 0, 930.00 W, 63, 1965 MHz, 3996 MHz, 99 %, 75 %', +].join('\n'); + +const AMD_HEADER = + 'timestamp,gpu,gfx_activity,umc_activity,mm_activity,socket_power,gfx_voltage,soc_voltage,mem_voltage,gfx_0_clk,mem_0_clk,fclk_0_clk,socclk_0_clk,edge,hotspot,mem'; + +const AMD_CSV = [ + AMD_HEADER, + `1789515524,0,0,0,N/A,288,N/A,N/A,N/A,2402,2000,1250,39,N/A,44,24`, + `1789515525,0,95,80,N/A,1180,850,900,1250,2100,2000,1250,39,55,78,60`, + `1789515524,1,0,0,N/A,290,N/A,N/A,N/A,2402,2000,1250,39,40,45,25`, +].join('\r\n'); + +describe('parseGpuMetricsCsv — NVIDIA', () => { + it('parses unit-suffixed cells into UTC epoch samples and drops rows without power', () => { + const parsed = parseGpuMetricsCsv(NVIDIA_CSV); + expect(parsed?.vendor).toBe('nvidia'); + expect(parsed?.samples).toHaveLength(4); + const first = parsed!.samples[0]!; + expect(first.timestampMs).toBe(Date.UTC(2026, 8, 11, 4, 19, 41, 982)); + expect(first.gpuIndex).toBe(0); + expect(first.powerW).toBe(187.8); + expect(first.temperatureC).toBe(33); + expect(first.smClockMhz).toBe(120); + expect(first.memClockMhz).toBe(3996); + expect(first.gpuUtilPct).toBe(0); + expect(first.memUtilPct).toBe(0); + expect(first.edgeTempC).toBeNull(); + // The `[N/A]` power row for GPU 1 is not a usable sample. + expect(parsed!.samples.filter((s) => s.gpuIndex === 1)).toHaveLength(1); + }); + + it('applies a fixed collector offset when the context is not UTC', () => { + const parsed = parseGpuMetricsCsv(NVIDIA_CSV, { nvidiaUtcOffsetMinutes: -300 }); + expect(parsed!.samples[0]!.timestampMs).toBe(Date.UTC(2026, 8, 11, 9, 19, 41, 982)); + }); + + it('returns null for a header-only file or an unknown header', () => { + expect(parseGpuMetricsCsv(NVIDIA_CSV.split('\n')[0]!)).toBeNull(); + expect(parseGpuMetricsCsv('a,b,c\n1,2,3')).toBeNull(); + }); +}); + +describe('parseGpuMetricsCsv — AMD', () => { + it('maps the amd-smi subset, preferring hotspot over edge temperature', () => { + const parsed = parseGpuMetricsCsv(AMD_CSV); + expect(parsed?.vendor).toBe('amd'); + expect(parsed?.samples).toHaveLength(3); + const idle = parsed!.samples[0]!; + expect(idle.timestampMs).toBe(1789515524000); + expect(idle.powerW).toBe(288); + expect(idle.temperatureC).toBe(44); + expect(idle.edgeTempC).toBeNull(); + expect(idle.memTempC).toBe(24); + expect(idle.gfxVoltageMv).toBeNull(); + const busy = parsed!.samples[1]!; + expect(busy.gpuUtilPct).toBe(95); + expect(busy.memUtilPct).toBe(80); + expect(busy.gfxVoltageMv).toBe(850); + expect(busy.fclkMhz).toBe(1250); + expect(busy.socclkMhz).toBe(39); + expect(busy.edgeTempC).toBe(55); + expect(busy.temperatureC).toBe(78); + }); +}); + +describe('timestamp and cell helpers', () => { + it('parses nvidia-smi timestamps with and without milliseconds', () => { + expect(parseNvidiaTimestamp('2026/01/02 03:04:05')).toBe(Date.UTC(2026, 0, 2, 3, 4, 5)); + expect(parseNvidiaTimestamp('2026/01/02 03:04:05.5')).toBe(Date.UTC(2026, 0, 2, 3, 4, 5, 500)); + expect(parseNvidiaTimestamp('not a date')).toBeNull(); + }); + + it('accepts epoch seconds, epoch milliseconds, and ISO strings for amd-smi', () => { + expect(parseAmdTimestamp('1789515524')).toBe(1789515524000); + expect(parseAmdTimestamp('1789515524.25')).toBe(1789515524250); + expect(parseAmdTimestamp('1789515524000')).toBe(1789515524000); + expect(parseAmdTimestamp('2026-09-16T00:00:00Z')).toBe(Date.UTC(2026, 8, 16)); + expect(parseAmdTimestamp('12')).toBeNull(); + }); + + it('treats N/A and blanks as null', () => { + expect(parseMetricCell('N/A')).toBeNull(); + expect(parseMetricCell('')).toBeNull(); + expect(parseMetricCell(undefined)).toBeNull(); + expect(parseMetricCell(' 12.5 W')).toBe(12.5); + }); +}); + +describe('computeGpuMetricStats', () => { + it('digests every non-null metric per GPU with interpolated percentiles', () => { + const parsed = parseGpuMetricsCsv(NVIDIA_CSV)!; + const stats = computeGpuMetricStats(parsed.samples); + const gpu0Power = stats.find((s) => s.gpuIndex === 0 && s.metric === 'powerW')!; + expect(gpu0Power.count).toBe(3); + expect(gpu0Power.min).toBe(187.8); + expect(gpu0Power.max).toBe(930); + expect(gpu0Power.mean).toBeCloseTo((187.8 + 912.1 + 930) / 3, 6); + expect(gpu0Power.median).toBe(912.1); + expect(gpu0Power.p95).toBeCloseTo(912.1 + (930 - 912.1) * 0.9, 6); + expect(gpu0Power.p99).toBeCloseTo(912.1 + (930 - 912.1) * 0.98, 6); + expect(gpu0Power.stddev).toBeGreaterThan(0); + // AMD-only columns never appear for an NVIDIA series. + expect(stats.some((s) => s.metric === 'edgeTempC')).toBe(false); + expect(stats.filter((s) => s.gpuIndex === 1 && s.metric === 'powerW')[0]!.count).toBe(1); + }); + + it('returns an empty digest for no samples', () => { + expect(computeGpuMetricStats([])).toEqual([]); + }); +}); + +describe('summarizeGpuMetricSamples', () => { + it('reports the window, GPU count, and median per-GPU cadence', () => { + const summary = summarizeGpuMetricSamples(parseGpuMetricsCsv(NVIDIA_CSV)!.samples)!; + expect(summary.sampleCount).toBe(4); + expect(summary.gpuCount).toBe(2); + expect(summary.startedAtMs).toBe(Date.UTC(2026, 8, 11, 4, 19, 41, 982)); + expect(summary.endedAtMs).toBe(Date.UTC(2026, 8, 11, 4, 19, 43, 990)); + expect(summary.sampleIntervalS).toBeCloseTo(1.004, 3); + }); + + it('returns null for an empty series', () => { + expect(summarizeGpuMetricSamples([])).toBeNull(); + }); +}); diff --git a/packages/db/src/etl/gpu-metrics-csv.ts b/packages/db/src/etl/gpu-metrics-csv.ts new file mode 100644 index 000000000..35ec33e57 --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-csv.ts @@ -0,0 +1,385 @@ +/** + * Parser and digest for the producer's `gpu_metrics.csv` telemetry. + * + * The InferenceX runner samples `nvidia-smi --query-gpu=…` or `amd-smi metric + * --csv` once per second for the lifetime of a benchmark job. This module is + * pure (no I/O) so the CI ingest, the historical backfill, and the app's + * GitHub fallback path all normalize the two vendor formats identically. + */ + +export type GpuMetricsVendor = 'nvidia' | 'amd'; + +export interface GpuMetricSample { + /** Sample instant in epoch milliseconds (UTC). */ + timestampMs: number; + gpuIndex: number; + powerW: number | null; + temperatureC: number | null; + smClockMhz: number | null; + memClockMhz: number | null; + gpuUtilPct: number | null; + memUtilPct: number | null; + edgeTempC: number | null; + memTempC: number | null; + gfxVoltageMv: number | null; + socVoltageMv: number | null; + memVoltageMv: number | null; + fclkMhz: number | null; + socclkMhz: number | null; + mmActivityPct: number | null; +} + +export interface ParsedGpuMetricsCsv { + vendor: GpuMetricsVendor; + samples: GpuMetricSample[]; +} + +/** Sample columns that participate in the per-GPU statistics digest. */ +export const GPU_METRIC_STAT_KEYS = [ + 'powerW', + 'temperatureC', + 'smClockMhz', + 'memClockMhz', + 'gpuUtilPct', + 'memUtilPct', + 'edgeTempC', + 'memTempC', + 'gfxVoltageMv', + 'socVoltageMv', + 'memVoltageMv', + 'fclkMhz', + 'socclkMhz', + 'mmActivityPct', +] as const; + +export type GpuMetricStatKey = (typeof GPU_METRIC_STAT_KEYS)[number]; + +export interface GpuMetricStats { + gpuIndex: number; + metric: GpuMetricStatKey; + count: number; + min: number; + max: number; + mean: number; + median: number; + p95: number; + p99: number; + stddev: number; +} + +export function splitCsvLine(line: string): string[] { + const result: string[] = []; + let current = ''; + let inQuotes = false; + for (const char of line) { + if (char === '"') { + inQuotes = !inQuotes; + } else if (char === ',' && !inQuotes) { + result.push(current.trim()); + current = ''; + } else { + current += char; + } + } + result.push(current.trim()); + return result; +} + +function buildColumnMap(headerLine: string): Map { + const headers = splitCsvLine(headerLine); + const map = new Map(); + for (let i = 0; i < headers.length; i++) map.set(headers[i]!.toLowerCase(), i); + return map; +} + +/** `N/A`, empty, and unparseable cells become null; unit suffixes are ignored. */ +export function parseMetricCell(value: string | undefined): number | null { + if (value === undefined) return null; + const trimmed = value.trim(); + if (trimmed === '' || trimmed === 'N/A' || trimmed === '[N/A]') return null; + const parsed = Number.parseFloat(trimmed); + return Number.isFinite(parsed) ? parsed : null; +} + +const NVIDIA_TIMESTAMP_RE = + /^(?\d{4})\/(?\d{2})\/(?\d{2}) (?\d{2}):(?\d{2}):(?\d{2})(?:\.(?\d{1,3}))?$/u; + +/** + * nvidia-smi prints `YYYY/MM/DD HH:MM:SS.mmm` in the collector's local zone. + * The producer runs its collectors with TZ=UTC; callers can pass a different + * fixed offset (minutes east of UTC) when a context sidecar says otherwise. + */ +export function parseNvidiaTimestamp(raw: string, offsetMinutes = 0): number | null { + const match = NVIDIA_TIMESTAMP_RE.exec(raw.trim()); + if (!match?.groups) return null; + const { y, mo, d, h, mi, s, ms } = match.groups; + const millis = ms ? Number(ms.padEnd(3, '0')) : 0; + const utc = Date.UTC( + Number(y), + Number(mo) - 1, + Number(d), + Number(h), + Number(mi), + Number(s), + millis, + ); + return utc - offsetMinutes * 60_000; +} + +/** amd-smi prints Unix epoch seconds; accept fractional and millisecond forms. */ +export function parseAmdTimestamp(raw: string): number | null { + const trimmed = raw.trim(); + if (/^\d+(?:\.\d+)?$/u.test(trimmed)) { + const numeric = Number.parseFloat(trimmed); + if (numeric > 1e12) return Math.round(numeric); // already milliseconds + if (numeric > 1e9) return Math.round(numeric * 1000); + return null; + } + const iso = Date.parse(trimmed); + return Number.isFinite(iso) ? iso : null; +} + +function emptySample(timestampMs: number, gpuIndex: number): GpuMetricSample { + return { + timestampMs, + gpuIndex, + powerW: null, + temperatureC: null, + smClockMhz: null, + memClockMhz: null, + gpuUtilPct: null, + memUtilPct: null, + edgeTempC: null, + memTempC: null, + gfxVoltageMv: null, + socVoltageMv: null, + memVoltageMv: null, + fclkMhz: null, + socclkMhz: null, + mmActivityPct: null, + }; +} + +/** + * Columns: timestamp, index, power.draw [W], temperature.gpu, + * clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], + * utilization.memory [%]. Header order is fixed by NVIDIA_GPU_MONITOR_QUERY in + * the producer, but we still resolve by name so a reordered query keeps working. + */ +function parseNvidiaCsv(lines: readonly string[], offsetMinutes: number): GpuMetricSample[] { + const header = splitCsvLine(lines[0]!).map((h) => h.toLowerCase().replace(/\s*\[.*\]$/u, '')); + const col = (name: string) => header.indexOf(name); + const iTimestamp = col('timestamp'); + const iIndex = col('index'); + const iPower = col('power.draw'); + const iTemp = col('temperature.gpu'); + const iSm = col('clocks.current.sm'); + const iMem = col('clocks.current.memory'); + const iUtil = col('utilization.gpu'); + const iMemUtil = col('utilization.memory'); + if (iTimestamp < 0 || iIndex < 0 || iPower < 0) return []; + + const samples: GpuMetricSample[] = []; + for (let i = 1; i < lines.length; i++) { + const cols = splitCsvLine(lines[i]!); + if (cols.length <= Math.max(iTimestamp, iIndex, iPower)) continue; + const timestampMs = parseNvidiaTimestamp(cols[iTimestamp]!, offsetMinutes); + const gpuIndex = Number.parseInt(cols[iIndex]!, 10); + if (timestampMs === null || !Number.isInteger(gpuIndex) || gpuIndex < 0) continue; + const sample = emptySample(timestampMs, gpuIndex); + sample.powerW = parseMetricCell(cols[iPower]); + if (sample.powerW === null) continue; + sample.temperatureC = iTemp >= 0 ? parseMetricCell(cols[iTemp]) : null; + sample.smClockMhz = iSm >= 0 ? parseMetricCell(cols[iSm]) : null; + sample.memClockMhz = iMem >= 0 ? parseMetricCell(cols[iMem]) : null; + sample.gpuUtilPct = iUtil >= 0 ? parseMetricCell(cols[iUtil]) : null; + sample.memUtilPct = iMemUtil >= 0 ? parseMetricCell(cols[iMemUtil]) : null; + samples.push(sample); + } + return samples; +} + +/** + * amd-smi `metric --csv` is a wide table. The consumed subset mirrors the + * dashboard: socket_power, gfx_activity, umc_activity, mm_activity, the first + * gfx/mem/fclk/socclk clock domains, hotspot/edge/mem temperatures and the + * three voltage rails. + */ +function parseAmdCsv(lines: readonly string[]): GpuMetricSample[] { + const colMap = buildColumnMap(lines[0]!); + const col = (name: string) => colMap.get(name) ?? -1; + const iTimestamp = col('timestamp'); + const iGpu = col('gpu'); + const iPower = col('socket_power'); + const iGfxActivity = col('gfx_activity'); + const iUmcActivity = col('umc_activity'); + const iGfxClk = col('gfx_0_clk'); + const iMemClk = col('mem_0_clk'); + const iHotspot = col('hotspot'); + const iEdge = col('edge'); + const iMemTemp = col('mem'); + const iGfxVoltage = col('gfx_voltage'); + const iSocVoltage = col('soc_voltage'); + const iMemVoltage = col('mem_voltage'); + const iFclk = col('fclk_0_clk'); + const iSocClk = col('socclk_0_clk'); + const iMmActivity = col('mm_activity'); + if (iTimestamp < 0 || iGpu < 0 || iPower < 0) return []; + + const samples: GpuMetricSample[] = []; + for (let i = 1; i < lines.length; i++) { + const cols = splitCsvLine(lines[i]!); + if (cols.length <= Math.max(iTimestamp, iGpu, iPower)) continue; + const timestampMs = parseAmdTimestamp(cols[iTimestamp]!); + const gpuIndex = Number.parseInt(cols[iGpu]!, 10); + if (timestampMs === null || !Number.isInteger(gpuIndex) || gpuIndex < 0) continue; + const sample = emptySample(timestampMs, gpuIndex); + sample.powerW = parseMetricCell(cols[iPower]); + if (sample.powerW === null) continue; + const hotspot = iHotspot >= 0 ? parseMetricCell(cols[iHotspot]) : null; + const edge = iEdge >= 0 ? parseMetricCell(cols[iEdge]) : null; + sample.temperatureC = hotspot ?? edge; + sample.edgeTempC = edge; + sample.memTempC = iMemTemp >= 0 ? parseMetricCell(cols[iMemTemp]) : null; + sample.smClockMhz = iGfxClk >= 0 ? parseMetricCell(cols[iGfxClk]) : null; + sample.memClockMhz = iMemClk >= 0 ? parseMetricCell(cols[iMemClk]) : null; + sample.gpuUtilPct = iGfxActivity >= 0 ? parseMetricCell(cols[iGfxActivity]) : null; + sample.memUtilPct = iUmcActivity >= 0 ? parseMetricCell(cols[iUmcActivity]) : null; + sample.gfxVoltageMv = iGfxVoltage >= 0 ? parseMetricCell(cols[iGfxVoltage]) : null; + sample.socVoltageMv = iSocVoltage >= 0 ? parseMetricCell(cols[iSocVoltage]) : null; + sample.memVoltageMv = iMemVoltage >= 0 ? parseMetricCell(cols[iMemVoltage]) : null; + sample.fclkMhz = iFclk >= 0 ? parseMetricCell(cols[iFclk]) : null; + sample.socclkMhz = iSocClk >= 0 ? parseMetricCell(cols[iSocClk]) : null; + sample.mmActivityPct = iMmActivity >= 0 ? parseMetricCell(cols[iMmActivity]) : null; + samples.push(sample); + } + return samples; +} + +export interface ParseGpuMetricsOptions { + /** Fixed offset of the NVIDIA collector clock, minutes east of UTC. */ + nvidiaUtcOffsetMinutes?: number; +} + +/** + * Parse one gpu_metrics CSV, auto-detecting the vendor from the header. + * Samples are returned in file order; callers sort per GPU as needed. + */ +export function parseGpuMetricsCsv( + csvText: string, + options: ParseGpuMetricsOptions = {}, +): ParsedGpuMetricsCsv | null { + const lines = csvText + .split('\n') + .map((line) => line.replace(/\r$/u, '')) + .filter((line) => line.trim().length > 0); + if (lines.length <= 1) return null; + const headerLower = lines[0]!.toLowerCase(); + if (headerLower.includes('socket_power') || headerLower.includes('gfx_activity')) { + return { vendor: 'amd', samples: parseAmdCsv(lines) }; + } + if (headerLower.includes('power.draw')) { + return { + vendor: 'nvidia', + samples: parseNvidiaCsv(lines, options.nvidiaUtcOffsetMinutes ?? 0), + }; + } + return null; +} + +function percentile(sorted: readonly number[], p: number): number { + const idx = (p / 100) * (sorted.length - 1); + const lo = Math.floor(idx); + const hi = Math.ceil(idx); + if (lo === hi) return sorted[lo]!; + return sorted[lo]! + (sorted[hi]! - sorted[lo]!) * (idx - lo); +} + +/** Per-GPU, per-metric summary over every non-null sample. */ +export function computeGpuMetricStats(samples: readonly GpuMetricSample[]): GpuMetricStats[] { + const groups = new Map< + string, + { gpuIndex: number; metric: GpuMetricStatKey; values: number[] } + >(); + for (const sample of samples) { + for (const metric of GPU_METRIC_STAT_KEYS) { + const value = sample[metric]; + if (value === null) continue; + const key = `${sample.gpuIndex}|${metric}`; + let group = groups.get(key); + if (!group) { + group = { gpuIndex: sample.gpuIndex, metric, values: [] }; + groups.set(key, group); + } + group.values.push(value); + } + } + + const stats: GpuMetricStats[] = []; + for (const group of groups.values()) { + const sorted = group.values.toSorted((a, b) => a - b); + const mean = sorted.reduce((acc, v) => acc + v, 0) / sorted.length; + const variance = sorted.reduce((acc, v) => acc + (v - mean) ** 2, 0) / sorted.length; + stats.push({ + gpuIndex: group.gpuIndex, + metric: group.metric, + count: sorted.length, + min: sorted[0]!, + max: sorted.at(-1)!, + mean, + median: percentile(sorted, 50), + p95: percentile(sorted, 95), + p99: percentile(sorted, 99), + stddev: Math.sqrt(variance), + }); + } + return stats.toSorted((a, b) => a.gpuIndex - b.gpuIndex || a.metric.localeCompare(b.metric)); +} + +export interface GpuMetricSeriesSummary { + sampleCount: number; + gpuCount: number; + startedAtMs: number; + endedAtMs: number; + /** Median gap between consecutive samples of one GPU, in seconds. */ + sampleIntervalS: number | null; +} + +/** Window and cadence facts stored on the series row. */ +export function summarizeGpuMetricSamples( + samples: readonly GpuMetricSample[], +): GpuMetricSeriesSummary | null { + if (samples.length === 0) return null; + let startedAtMs = Number.POSITIVE_INFINITY; + let endedAtMs = Number.NEGATIVE_INFINITY; + const byGpu = new Map(); + for (const sample of samples) { + if (sample.timestampMs < startedAtMs) startedAtMs = sample.timestampMs; + if (sample.timestampMs > endedAtMs) endedAtMs = sample.timestampMs; + const bucket = byGpu.get(sample.gpuIndex); + if (bucket) bucket.push(sample.timestampMs); + else byGpu.set(sample.gpuIndex, [sample.timestampMs]); + } + const gaps: number[] = []; + for (const timestamps of byGpu.values()) { + const sorted = timestamps.toSorted((a, b) => a - b); + for (let i = 1; i < sorted.length; i++) { + const gap = sorted[i]! - sorted[i - 1]!; + if (gap > 0) gaps.push(gap); + } + } + const sampleIntervalS = + gaps.length > 0 + ? percentile( + gaps.toSorted((a, b) => a - b), + 50, + ) / 1000 + : null; + return { + sampleCount: samples.length, + gpuCount: byGpu.size, + startedAtMs, + endedAtMs, + sampleIntervalS, + }; +} diff --git a/packages/db/src/etl/gpu-metrics-ingest.test.ts b/packages/db/src/etl/gpu-metrics-ingest.test.ts new file mode 100644 index 000000000..c1d135aa5 --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-ingest.test.ts @@ -0,0 +1,203 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { PGlite } from '@electric-sql/pglite'; +import type postgres from 'postgres'; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from 'vitest'; + +import { ingestGpuMetricsArtifact, prepareGpuMetricsArtifact } from './gpu-metrics-ingest'; + +type Sql = postgres.Sql; +let db: PGlite; +let sql: Sql; +const roots: string[] = []; + +function queryClient(database: Pick) { + const client = async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await database.query(query, values); + return result.rows; + }; + return Object.assign(client, { + json: JSON.stringify, + array: (value: unknown) => value, + }); +} + +beforeAll(async () => { + db = await PGlite.create(); + for (const name of ['001_initial_schema.sql', '016_gpu_metrics.sql']) { + await db.exec(fs.readFileSync(new URL(`../../migrations/${name}`, import.meta.url), 'utf8')); + } + sql = Object.assign(queryClient(db), { + begin: (fn: (tx: Sql) => Promise) => + db.transaction((tx) => fn(queryClient(tx) as unknown as Sql)), + }) as unknown as Sql; +}, 20_000); + +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, status, conclusion, created_at, date) + VALUES (1, 34557177019, 1, 'Run Sweep', 'completed', 'success', '2026-09-11', '2026-09-11'); + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, 4, 4, 4, 4); + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, isl, osl, conc, metrics) + VALUES (10, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, '{}'), + (11, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 64, '{}');`); +}); + +afterEach(() => { + for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); +}); + +afterAll(async () => { + await db?.close(); +}); + +const NVIDIA_CSV = [ + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]', + '2026/09/11 04:19:41.982, 0, 187.80 W, 33, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:41.986, 1, 190.96 W, 39, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:42.990, 0, 912.10 W, 61, 1965 MHz, 3996 MHz, 98 %, 74 %', + '2026/09/11 04:19:42.994, 1, 905.30 W, 62, 1965 MHz, 3996 MHz, 97 %, 73 %', + // Duplicate flush of the last sample, as emitted when the monitor stops. + '2026/09/11 04:19:42.994, 1, 905.30 W, 62, 1965 MHz, 3996 MHz, 97 %, 73 %', +].join('\n'); + +function writeArtifact(csv: string, contextZone = 'UTC') { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gpu-metrics-ingest-')); + roots.push(root); + const artifactName = 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0'; + const artifactDir = path.join(root, artifactName); + fs.mkdirSync(artifactDir); + fs.writeFileSync(path.join(artifactDir, 'gpu_metrics.csv'), csv); + fs.writeFileSync( + path.join(artifactDir, 'gpu_metrics_context.json'), + JSON.stringify({ timestamp_timezone: contextZone }), + ); + fs.writeFileSync( + path.join(artifactDir, 'gpu_metrics_identity.csv'), + 'index, uuid, name\n0, GPU-a, NVIDIA B200\n1, GPU-b, NVIDIA B200\n', + ); + return { artifactName, artifactDir }; +} + +describe('prepareGpuMetricsArtifact', () => { + it('parses each CSV with its sidecars and digest', () => { + const [series] = prepareGpuMetricsArtifact(writeArtifact(NVIDIA_CSV)); + expect(series?.fileName).toBe('gpu_metrics.csv'); + expect(series?.vendor).toBe('nvidia'); + expect(series?.samples).toHaveLength(5); + expect(series?.gpuCount).toBe(2); + expect(series?.sidecars.context).toEqual({ timestamp_timezone: 'UTC' }); + expect(series?.stats.find((s) => s.gpuIndex === 0 && s.metric === 'powerW')?.max).toBe(912.1); + }); +}); + +describe('ingestGpuMetricsArtifact', () => { + it('stores series, samples, digest, and point links; reruns are no-ops', async () => { + const artifact = writeArtifact(NVIDIA_CSV); + const first = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10, 10], + }); + expect(first.seriesIds).toHaveLength(1); + // Five CSV rows, but the repeated final sample collapses on the primary key. + expect(first.samplesInserted).toBe(4); + expect(first.seriesSkipped).toBe(0); + + const [series] = await sql< + { + artifact_name: string; + config_key: string; + vendor: string; + sample_count: number; + gpu_count: number; + started_at: Date; + ended_at: Date; + sample_interval_s: number | null; + }[] + >`select artifact_name, config_key, vendor, sample_count, gpu_count, started_at, ended_at, + sample_interval_s from gpu_metric_series`; + expect(series!.artifact_name).toBe(artifact.artifactName); + expect(series!.config_key).toBe('dsr1_8k1k_fp4_sglang_conc32_b200-x_0'); + expect(series!.vendor).toBe('nvidia'); + expect(series!.sample_count).toBe(5); + expect(series!.gpu_count).toBe(2); + expect(new Date(series!.started_at).toISOString()).toBe('2026-09-11T04:19:41.982Z'); + expect(new Date(series!.ended_at).toISOString()).toBe('2026-09-11T04:19:42.994Z'); + expect(series!.sample_interval_s).toBeCloseTo(1.008, 3); + + const samples = await sql<{ gpu_index: number; power_w: number; sampled_at: Date }[]>` + select gpu_index, power_w, sampled_at from gpu_metric_samples order by sampled_at, gpu_index`; + expect(samples.map((s) => [s.gpu_index, s.power_w])).toEqual([ + [0, 187.8], + [1, 190.96], + [0, 912.1], + [1, 905.3], + ]); + + const stats = await sql< + { gpu_index: number; metric: string; sample_count: number; max_value: number }[] + >` + select gpu_index, metric, sample_count, max_value from gpu_metric_gpu_stats + where metric = 'power_w' order by gpu_index`; + expect(stats).toEqual([ + { gpu_index: 0, metric: 'power_w', sample_count: 2, max_value: 912.1 }, + { gpu_index: 1, metric: 'power_w', sample_count: 3, max_value: 905.3 }, + ]); + + const links = await sql<{ benchmark_result_id: number }[]>` + select benchmark_result_id from benchmark_result_gpu_metrics order by 1`; + expect(links.map((l) => Number(l.benchmark_result_id))).toEqual([10]); + + const rerun = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10, 11], + }); + expect(rerun.seriesIds).toEqual(first.seriesIds); + expect(rerun.samplesInserted).toBe(0); + expect(rerun.seriesSkipped).toBe(1); + const relinked = await sql<{ benchmark_result_id: number }[]>` + select benchmark_result_id from benchmark_result_gpu_metrics order by 1`; + expect(relinked.map((l) => Number(l.benchmark_result_id))).toEqual([10, 11]); + const [count] = await sql<{ n: number }[]>`select count(*)::int as n from gpu_metric_samples`; + expect(count!.n).toBe(4); + }); + + it('replaces samples and digest when the CSV content changes', async () => { + const artifact = writeArtifact(NVIDIA_CSV); + await ingestGpuMetricsArtifact(sql, { workflowRunId: 1, artifact, benchmarkResultIds: [10] }); + + const longer = `${NVIDIA_CSV}\n2026/09/11 04:19:43.990, 0, 950.00 W, 65, 1965 MHz, 3996 MHz, 99 %, 75 %`; + fs.writeFileSync(path.join(artifact.artifactDir, 'gpu_metrics.csv'), longer); + const result = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10], + }); + expect(result.samplesInserted).toBe(5); + expect(result.seriesSkipped).toBe(0); + const [series] = await sql<{ n: number; sample_count: number }[]>` + select (select count(*)::int from gpu_metric_series) as n, sample_count from gpu_metric_series`; + expect(series!.n).toBe(1); + expect(series!.sample_count).toBe(6); + const [stat] = await sql<{ max_value: number }[]>` + select max_value from gpu_metric_gpu_stats where gpu_index = 0 and metric = 'power_w'`; + expect(stat!.max_value).toBe(950); + }); + + it('ignores artifacts whose CSVs cannot be parsed', async () => { + const artifact = writeArtifact('nonsense,header\n1,2\n'); + const result = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10], + }); + expect(result).toEqual({ seriesIds: [], samplesInserted: 0, seriesSkipped: 0 }); + }); +}); diff --git a/packages/db/src/etl/gpu-metrics-ingest.ts b/packages/db/src/etl/gpu-metrics-ingest.ts new file mode 100644 index 000000000..dc68d0e74 --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-ingest.ts @@ -0,0 +1,297 @@ +/** + * Persist one gpu_metrics artifact: series metadata, full-resolution samples, + * the per-GPU statistics digest, and links to the benchmark points it covers. + * + * Idempotency: a series is identified by (workflow run, artifact name, CSV + * path). Re-ingesting an identical CSV (same sha256) only refreshes the point + * links; a changed CSV replaces the stored samples and digest inside one + * transaction so readers never observe a half-written series. + */ + +import { createHash } from 'node:crypto'; +import fs from 'node:fs'; + +import type postgres from 'postgres'; + +import type { Sql } from './db-utils.js'; + +/** Either a pooled client or the transaction handle passed to `sql.begin` callbacks. */ +type TxLike = Sql | postgres.TransactionSql; +import { + computeGpuMetricStats, + parseGpuMetricsCsv, + summarizeGpuMetricSamples, + type GpuMetricSample, + type GpuMetricStats, + type GpuMetricsVendor, +} from './gpu-metrics-csv.js'; +import { + contextUtcOffsetMinutes, + gpuMetricsArtifactSuffix, + listGpuMetricsCsvFiles, + readGpuMetricsSidecars, + type GpuMetricsArtifact, + type GpuMetricsSidecars, +} from './gpu-metrics-artifacts.js'; + +/** Samples are streamed to Postgres in unnest batches of this many rows. */ +const SAMPLE_BATCH_SIZE = 5000; + +export interface PreparedGpuMetricSeries { + fileName: string; + vendor: GpuMetricsVendor; + csvSha256: string; + samples: GpuMetricSample[]; + stats: GpuMetricStats[]; + sampleIntervalS: number | null; + gpuCount: number; + startedAtMs: number; + endedAtMs: number; + sidecars: GpuMetricsSidecars; +} + +export interface GpuMetricsIngestResult { + seriesIds: number[]; + samplesInserted: number; + seriesSkipped: number; +} + +/** Parse every CSV in an extracted artifact; unparseable files are skipped. */ +export function prepareGpuMetricsArtifact(artifact: GpuMetricsArtifact): PreparedGpuMetricSeries[] { + const prepared: PreparedGpuMetricSeries[] = []; + for (const file of listGpuMetricsCsvFiles(artifact.artifactDir)) { + const csvText = fs.readFileSync(file.path, 'utf8'); + const sidecars = readGpuMetricsSidecars(file.path); + const parsed = parseGpuMetricsCsv(csvText, { + nvidiaUtcOffsetMinutes: contextUtcOffsetMinutes(sidecars.context), + }); + if (!parsed) continue; + const summary = summarizeGpuMetricSamples(parsed.samples); + if (!summary) continue; + prepared.push({ + fileName: file.fileName, + vendor: parsed.vendor, + csvSha256: createHash('sha256').update(csvText).digest('hex'), + samples: parsed.samples, + stats: computeGpuMetricStats(parsed.samples), + sampleIntervalS: summary.sampleIntervalS, + gpuCount: summary.gpuCount, + startedAtMs: summary.startedAtMs, + endedAtMs: summary.endedAtMs, + sidecars, + }); + } + return prepared; +} + +const STAT_METRIC_COLUMN: Record = { + powerW: 'power_w', + temperatureC: 'temperature_c', + smClockMhz: 'sm_clock_mhz', + memClockMhz: 'mem_clock_mhz', + gpuUtilPct: 'gpu_util_pct', + memUtilPct: 'mem_util_pct', + edgeTempC: 'edge_temp_c', + memTempC: 'mem_temp_c', + gfxVoltageMv: 'gfx_voltage_mv', + socVoltageMv: 'soc_voltage_mv', + memVoltageMv: 'mem_voltage_mv', + fclkMhz: 'fclk_mhz', + socclkMhz: 'socclk_mhz', + mmActivityPct: 'mm_activity_pct', +}; + +/** Stored metric names use the column spelling so SQL readers need no mapping. */ +export function statMetricColumn(metric: GpuMetricStats['metric']): string { + return STAT_METRIC_COLUMN[metric]; +} + +function nullable(values: (number | null)[]): (number | null)[] { + return values; +} + +async function insertSampleBatch( + tx: TxLike, + seriesId: number, + batch: readonly GpuMetricSample[], +): Promise { + const inserted = await tx<{ n: number }[]>` + with ins as ( + insert into gpu_metric_samples ( + series_id, gpu_index, sampled_at, + power_w, temperature_c, sm_clock_mhz, mem_clock_mhz, gpu_util_pct, mem_util_pct, + edge_temp_c, mem_temp_c, gfx_voltage_mv, soc_voltage_mv, mem_voltage_mv, + fclk_mhz, socclk_mhz, mm_activity_pct + ) + select + ${seriesId}, + unnest(${tx.array(batch.map((s) => s.gpuIndex))}::smallint[]), + to_timestamp(unnest(${tx.array(batch.map((s) => s.timestampMs / 1000))}::double precision[])), + unnest(${tx.array(nullable(batch.map((s) => s.powerW)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.temperatureC)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.smClockMhz)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.memClockMhz)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.gpuUtilPct)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.memUtilPct)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.edgeTempC)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.memTempC)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.gfxVoltageMv)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.socVoltageMv)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.memVoltageMv)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.fclkMhz)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.socclkMhz)))}::real[]), + unnest(${tx.array(nullable(batch.map((s) => s.mmActivityPct)))}::real[]) + -- nvidia-smi occasionally repeats the final sample when the monitor is + -- stopped and flushed; the primary key makes that a no-op. + on conflict (series_id, gpu_index, sampled_at) do nothing + returning 1 + ) + select count(*)::int as n from ins + `; + return Number(inserted[0]?.n ?? 0); +} + +async function insertStats( + tx: TxLike, + seriesId: number, + stats: readonly GpuMetricStats[], +): Promise { + if (stats.length === 0) return; + await tx` + insert into gpu_metric_gpu_stats ( + series_id, gpu_index, metric, sample_count, + min_value, max_value, mean_value, median_value, p95_value, p99_value, stddev_value + ) + select + ${seriesId}, + unnest(${tx.array(stats.map((s) => s.gpuIndex))}::smallint[]), + unnest(${tx.array(stats.map((s) => statMetricColumn(s.metric)))}::text[]), + unnest(${tx.array(stats.map((s) => s.count))}::int[]), + unnest(${tx.array(stats.map((s) => s.min))}::real[]), + unnest(${tx.array(stats.map((s) => s.max))}::real[]), + unnest(${tx.array(stats.map((s) => s.mean))}::real[]), + unnest(${tx.array(stats.map((s) => s.median))}::real[]), + unnest(${tx.array(stats.map((s) => s.p95))}::real[]), + unnest(${tx.array(stats.map((s) => s.p99))}::real[]), + unnest(${tx.array(stats.map((s) => s.stddev))}::real[]) + `; +} + +/** + * Upsert one prepared series and link it to `benchmarkResultIds`. Returns the + * series id and how many sample rows were written (0 when the CSV was already + * stored with the same hash). + */ +export function upsertGpuMetricSeries( + sql: Sql, + input: { + workflowRunId: number; + artifactName: string; + series: PreparedGpuMetricSeries; + benchmarkResultIds: readonly number[]; + }, +): Promise<{ seriesId: number; samplesInserted: number; replaced: boolean }> { + const { workflowRunId, artifactName, series, benchmarkResultIds } = input; + const configKey = gpuMetricsArtifactSuffix(artifactName) ?? artifactName; + const sidecarsJson = JSON.stringify(series.sidecars); + + return sql.begin(async (tx) => { + const existing = await tx<{ id: number; csv_sha256: string }[]>` + select id, csv_sha256 from gpu_metric_series + where workflow_run_id = ${workflowRunId} + and artifact_name = ${artifactName} + and file_name = ${series.fileName} + for update + `; + + let seriesId: number; + let needsSamples = true; + let replaced = false; + if (existing.length > 0) { + seriesId = Number(existing[0]!.id); + if (existing[0]!.csv_sha256 === series.csvSha256) { + needsSamples = false; + } else { + replaced = true; + await tx`delete from gpu_metric_samples where series_id = ${seriesId}`; + await tx`delete from gpu_metric_gpu_stats where series_id = ${seriesId}`; + await tx` + update gpu_metric_series set + vendor = ${series.vendor}, + csv_sha256 = ${series.csvSha256}, + sample_interval_s = ${series.sampleIntervalS}, + sample_count = ${series.samples.length}, + gpu_count = ${series.gpuCount}, + started_at = to_timestamp(${series.startedAtMs / 1000}::double precision), + ended_at = to_timestamp(${series.endedAtMs / 1000}::double precision), + sidecars = ${sidecarsJson}::jsonb, + ingested_at = now() + where id = ${seriesId} + `; + } + } else { + const [row] = await tx<{ id: number }[]>` + insert into gpu_metric_series ( + workflow_run_id, artifact_name, config_key, file_name, vendor, csv_sha256, + sample_interval_s, sample_count, gpu_count, started_at, ended_at, sidecars + ) values ( + ${workflowRunId}, ${artifactName}, ${configKey}, ${series.fileName}, + ${series.vendor}, ${series.csvSha256}, ${series.sampleIntervalS}, + ${series.samples.length}, ${series.gpuCount}, + to_timestamp(${series.startedAtMs / 1000}::double precision), + to_timestamp(${series.endedAtMs / 1000}::double precision), + ${sidecarsJson}::jsonb + ) + returning id + `; + seriesId = Number(row!.id); + } + + let samplesInserted = 0; + if (needsSamples) { + for (let offset = 0; offset < series.samples.length; offset += SAMPLE_BATCH_SIZE) { + samplesInserted += await insertSampleBatch( + tx, + seriesId, + series.samples.slice(offset, offset + SAMPLE_BATCH_SIZE), + ); + } + await insertStats(tx, seriesId, series.stats); + } + + if (benchmarkResultIds.length > 0) { + await tx` + insert into benchmark_result_gpu_metrics (benchmark_result_id, series_id) + select unnest(${tx.array([...new Set(benchmarkResultIds)])}::bigint[]), ${seriesId} + on conflict do nothing + `; + } + + return { seriesId, samplesInserted, replaced }; + }); +} + +/** Read, digest, and persist every CSV of one artifact for one set of points. */ +export async function ingestGpuMetricsArtifact( + sql: Sql, + input: { + workflowRunId: number; + artifact: GpuMetricsArtifact; + benchmarkResultIds: readonly number[]; + }, +): Promise { + const prepared = prepareGpuMetricsArtifact(input.artifact); + const result: GpuMetricsIngestResult = { seriesIds: [], samplesInserted: 0, seriesSkipped: 0 }; + for (const series of prepared) { + const upserted = await upsertGpuMetricSeries(sql, { + workflowRunId: input.workflowRunId, + artifactName: input.artifact.artifactName, + series, + benchmarkResultIds: input.benchmarkResultIds, + }); + result.seriesIds.push(upserted.seriesId); + result.samplesInserted += upserted.samplesInserted; + if (upserted.samplesInserted === 0 && !upserted.replaced) result.seriesSkipped++; + } + return result; +} diff --git a/packages/db/src/ingest-ci-run.ts b/packages/db/src/ingest-ci-run.ts index a27225bc9..90e87c33f 100644 --- a/packages/db/src/ingest-ci-run.ts +++ b/packages/db/src/ingest-ci-run.ts @@ -78,6 +78,8 @@ import { import { AsyncSemaphore } from './etl/async-semaphore'; import { discoverTraceReplayArtifacts } from './etl/trace-artifact-discovery'; import { discoverServerLogArtifacts, readServerLogArtifact } from './etl/server-log-artifacts'; +import { discoverGpuMetricsArtifacts } from './etl/gpu-metrics-artifacts'; +import { ingestGpuMetricsArtifact } from './etl/gpu-metrics-ingest'; import { datasetSlugFromBenchmarkRow } from './etl/dataset-provenance'; import { mapAggEvalRow, mapEvalRow } from './etl/eval-mapper'; import { ingestEvalRow } from './etl/eval-ingest'; @@ -452,6 +454,8 @@ async function main(): Promise { let totalSampleFiles = 0; let totalChangelogs = 0; let totalTraceReplayLinked = 0; + let totalGpuMetricSeries = 0; + let totalGpuMetricSamples = 0; const datasetSlugs = new Set(); // Dataset slugs referenced by this run's agentic rows but absent from the // `datasets` table — timeline→dataset deep links 404 until they're ingested. @@ -484,6 +488,13 @@ async function main(): Promise { if (serverLogArtifacts.size > 0) { console.log(` Found ${serverLogArtifacts.size} server log artifact(s)`); } + // PowerX telemetry: `gpu_metrics_` is uploaded next to `bmk_` by + // every benchmark job (see migration 016). Digested here so the dashboard + // never re-downloads GitHub artifacts and keeps the series past retention. + const gpuMetricsArtifacts = discoverGpuMetricsArtifacts(artifactsDir); + if (gpuMetricsArtifacts.size > 0) { + console.log(` Found ${gpuMetricsArtifacts.size} gpu_metrics artifact(s)`); + } // Sibling aiperf artifacts: each `bmk_agentic_` is paired with an // `agentic_` dir holding `profile_export.jsonl` and @@ -665,6 +676,30 @@ async function main(): Promise { tracker.recordDbError(`server_logs for ${configKey}`, error); } } + // Same pairing rule as server logs: `gpu_metrics_` carries no + // `agentic_` prefix, so agentic points fall back to the bare suffix. + const gpuMetricsArtifact = + gpuMetricsArtifacts.get(configKey) ?? + gpuMetricsArtifacts.get(stripBmkAndAgenticPrefix(parentDir)); + if (gpuMetricsArtifact) { + try { + const gpuMetricsStart = Date.now(); + const ingested = await ingestGpuMetricsArtifact(sql, { + workflowRunId, + artifact: gpuMetricsArtifact, + benchmarkResultIds: insertedIds, + }); + totalGpuMetricSeries += ingested.seriesIds.length; + totalGpuMetricSamples += ingested.samplesInserted; + console.log( + ` gpu_metrics ${ingested.seriesIds.length} series, ` + + `+${ingested.samplesInserted} sample(s), ` + + `${ingested.seriesSkipped} unchanged (${elapsed(gpuMetricsStart)})`, + ); + } catch (error: any) { + tracker.recordDbError(`gpu_metrics for ${configKey}`, error); + } + } } // Trace-replay sibling lookup for agentic points only. The aiperf diff --git a/packages/db/src/lib/benchmark-result-lookup.ts b/packages/db/src/lib/benchmark-result-lookup.ts new file mode 100644 index 000000000..37afe2acd --- /dev/null +++ b/packages/db/src/lib/benchmark-result-lookup.ts @@ -0,0 +1,97 @@ +/** + * Resolve persisted `benchmark_results` ids for raw artifact rows that were + * mapped through the production mapper. Shared by the sidecar backfills + * (server logs, gpu_metrics) so every historical attachment uses the same + * natural-key match as the CI ingest path. + */ + +import fs from 'node:fs'; +import path from 'node:path'; + +import { mapBenchmarkRow, type BenchmarkParams } from '../etl/benchmark-mapper.js'; +import { createSkipTracker } from '../etl/skip-tracker.js'; +import type { Sql } from '../etl/db-utils.js'; +import { resolveServerLogResultCandidates } from './server-log-backfill.js'; + +export interface BenchmarkRunSelector { + github_run_id: number; + run_attempt: number; +} + +export async function findBenchmarkResultIds( + sql: Sql, + run: BenchmarkRunSelector, + rows: readonly BenchmarkParams[], + onUniqueFallback: (id: number) => void = () => {}, +): Promise { + const ids = new Set(); + for (const row of rows) { + const c = row.config; + const candidates = await sql<{ id: number; offload_mode: string }[]>` + select br.id, br.offload_mode + from benchmark_results br + join workflow_runs wr on wr.id = br.workflow_run_id + join configs cfg on cfg.id = br.config_id + where wr.github_run_id = ${run.github_run_id} + and wr.run_attempt = ${run.run_attempt} + and cfg.hardware = ${c.hardware} + and cfg.framework = ${c.framework} + and cfg.model = ${c.model} + and cfg.precision = ${c.precision} + and cfg.spec_method = ${c.specMethod} + and cfg.disagg = ${c.disagg} + and cfg.is_multinode = ${c.isMultinode} + and cfg.prefill_tp = ${c.prefillTp} + and cfg.prefill_ep = ${c.prefillEp} + and cfg.prefill_dp_attention = ${c.prefillDpAttn} + and cfg.prefill_num_workers = ${c.prefillNumWorkers} + and cfg.decode_tp = ${c.decodeTp} + and cfg.decode_ep = ${c.decodeEp} + and cfg.decode_dp_attention = ${c.decodeDpAttn} + and cfg.decode_num_workers = ${c.decodeNumWorkers} + and cfg.num_prefill_gpu = ${c.numPrefillGpu} + and cfg.num_decode_gpu = ${c.numDecodeGpu} + and br.benchmark_type = ${row.benchmarkType} + and br.isl is not distinct from ${row.isl} + and br.osl is not distinct from ${row.osl} + and br.conc = ${row.conc} + and br.recipe_fingerprint is not distinct from ${row.recipeFingerprint} + `; + const resolution = resolveServerLogResultCandidates( + candidates.map((candidate) => ({ + id: Number(candidate.id), + offloadMode: candidate.offload_mode, + })), + row.offloadMode, + ); + if (resolution.usedUniqueFallback) onUniqueFallback(resolution.ids[0]!); + for (const id of resolution.ids) ids.add(id); + } + return [...ids]; +} + +function findJsonFiles(root: string): string[] { + const files: string[] = []; + for (const entry of fs.readdirSync(root, { withFileTypes: true })) { + const pathname = path.join(root, entry.name); + if (entry.isDirectory()) files.push(...findJsonFiles(pathname)); + else if (entry.isFile() && entry.name.endsWith('.json')) files.push(pathname); + } + return files.toSorted(); +} + +/** Map every raw benchmark JSON under an extracted bmk artifact through the production mapper. */ +export function readMappedBenchmarkRows(root: string): BenchmarkParams[] { + const tracker = createSkipTracker(); + const rows: BenchmarkParams[] = []; + for (const file of findJsonFiles(root)) { + const parsed = JSON.parse(fs.readFileSync(file, 'utf8')) as unknown; + const rawRows = Array.isArray(parsed) ? parsed : [parsed]; + for (const raw of rawRows) { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) continue; + const mapped = mapBenchmarkRow(raw as Record, tracker); + if (mapped) rows.push(mapped); + } + } + return rows; +} diff --git a/packages/db/src/lib/gpu-metrics-backfill.test.ts b/packages/db/src/lib/gpu-metrics-backfill.test.ts new file mode 100644 index 000000000..7ebf88d66 --- /dev/null +++ b/packages/db/src/lib/gpu-metrics-backfill.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from 'vitest'; + +import type { ArtifactMeta } from './github-artifacts.js'; +import { pairGpuMetricsArtifacts } from './gpu-metrics-backfill.js'; + +const meta = ( + name: string, + id: number, + created_at = '2026-09-11T00:00:00Z', + expired = false, +): ArtifactMeta => ({ + id, + name, + created_at, + expired, + archive_download_url: `https://example.test/${id}`, +}); + +describe('pairGpuMetricsArtifacts', () => { + it('pairs gpu_metrics with the exact bmk sibling and prefers the agentic sibling', () => { + const pairs = pairGpuMetricsArtifacts([ + meta('gpu_metrics_cfg-a_h200-cw_0', 1), + meta('bmk_cfg-a_h200-cw_0', 2), + meta('gpu_metrics_cfg-b_h200-cw_0', 3), + meta('bmk_agentic_cfg-b_h200-cw_0', 4), + meta('bmk_cfg-b_h200-cw_0', 5), + meta('gpu_metrics_orphan_h200-cw_0', 6), + meta('eval_gpu_metrics_cfg-a_h200-cw_0', 7), + ]); + expect(pairs.map((pair) => [pair.gpuMetrics.name, pair.benchmarks.name])).toEqual([ + ['gpu_metrics_cfg-a_h200-cw_0', 'bmk_cfg-a_h200-cw_0'], + ['gpu_metrics_cfg-b_h200-cw_0', 'bmk_agentic_cfg-b_h200-cw_0'], + ]); + }); + + it('keeps only the newest retry per logical benchmark and drops expired uploads', () => { + const pairs = pairGpuMetricsArtifacts([ + meta('gpu_metrics_cfg-a_h200-cw_0', 1, '2026-09-11T00:00:00Z'), + meta('bmk_cfg-a_h200-cw_0', 2, '2026-09-11T00:00:00Z'), + meta('gpu_metrics_cfg-a_h200-dgxc-slurm_1', 3, '2026-09-11T02:00:00Z'), + meta('bmk_cfg-a_h200-dgxc-slurm_1', 4, '2026-09-11T02:00:00Z'), + meta('gpu_metrics_cfg-c_h200-cw_0', 5, '2026-09-11T03:00:00Z', true), + meta('bmk_cfg-c_h200-cw_0', 6, '2026-09-11T03:00:00Z'), + ]); + expect(pairs.map((pair) => pair.gpuMetrics.name)).toEqual([ + 'gpu_metrics_cfg-a_h200-dgxc-slurm_1', + ]); + }); +}); diff --git a/packages/db/src/lib/gpu-metrics-backfill.ts b/packages/db/src/lib/gpu-metrics-backfill.ts new file mode 100644 index 000000000..94a09105a --- /dev/null +++ b/packages/db/src/lib/gpu-metrics-backfill.ts @@ -0,0 +1,47 @@ +/** Pairing rules for the historical gpu_metrics backfill. */ + +import { gpuMetricsArtifactSuffix } from '../etl/gpu-metrics-artifacts.js'; +import { RUNNER_SUFFIX_RE, type ArtifactMeta } from './github-artifacts.js'; + +export interface GpuMetricsArtifactPair { + gpuMetrics: ArtifactMeta; + benchmarks: ArtifactMeta; +} + +function isNewerArtifact(candidate: ArtifactMeta, existing: ArtifactMeta): boolean { + return ( + candidate.created_at > existing.created_at || + (candidate.created_at === existing.created_at && (candidate.id ?? 0) > (existing.id ?? 0)) + ); +} + +/** + * Pair every unexpired gpu_metrics artifact with its exact bmk sibling. Retried + * jobs upload on different runners, so the newest artifact per logical + * (runner-suffix-stripped) benchmark name wins, matching CI ingest. + */ +export function pairGpuMetricsArtifacts( + artifacts: readonly ArtifactMeta[], +): GpuMetricsArtifactPair[] { + const byName = new Map(); + for (const artifact of artifacts) { + if (artifact.expired) continue; + const existing = byName.get(artifact.name); + if (!existing || isNewerArtifact(artifact, existing)) byName.set(artifact.name, artifact); + } + const byLogicalBenchmark = new Map(); + for (const gpuMetrics of byName.values()) { + const suffix = gpuMetricsArtifactSuffix(gpuMetrics.name); + if (!suffix) continue; + const benchmarks = byName.get(`bmk_agentic_${suffix}`) ?? byName.get(`bmk_${suffix}`); + if (!benchmarks) continue; + const logicalName = benchmarks.name.replace(RUNNER_SUFFIX_RE, ''); + const existing = byLogicalBenchmark.get(logicalName); + if (!existing || isNewerArtifact(benchmarks, existing.benchmarks)) { + byLogicalBenchmark.set(logicalName, { gpuMetrics, benchmarks }); + } + } + return [...byLogicalBenchmark.values()].toSorted((a, b) => + a.gpuMetrics.name.localeCompare(b.gpuMetrics.name), + ); +} diff --git a/packages/db/src/queries/gpu-metrics.ts b/packages/db/src/queries/gpu-metrics.ts new file mode 100644 index 000000000..1d6bfdb64 --- /dev/null +++ b/packages/db/src/queries/gpu-metrics.ts @@ -0,0 +1,342 @@ +/** + * Read side of the PowerX telemetry digest (migration 016). + * + * Two entry points: everything recorded during one GitHub Actions run (the + * PowerX explorer keyed by run ID), and the series linked to one benchmark + * point (the per-point detail tab). Samples are returned as flat rows with + * ISO timestamps so the existing D3 charts consume them unchanged. + */ + +import type { DbClient } from '../connection.js'; + +export interface GpuMetricSampleRow { + timestamp: string; + index: number; + power: number; + temperature: number; + smClock: number; + memClock: number; + gpuUtil: number; + memUtil: number; + edgeTemp?: number; + memTemp?: number; + gfxVoltage?: number; + socVoltage?: number; + memVoltage?: number; + fclk?: number; + socClk?: number; + mmActivity?: number; +} + +export interface GpuMetricStatRow { + gpuIndex: number; + metric: string; + count: number; + min: number; + max: number; + mean: number; + median: number; + p95: number; + p99: number; + stddev: number; +} + +export interface GpuMetricSeries { + id: number; + artifactName: string; + configKey: string; + fileName: string; + vendor: string; + sampleIntervalS: number | null; + sampleCount: number; + gpuCount: number; + startedAt: string; + endedAt: string; + sidecars: Record; + benchmarkResultIds: number[]; + stats: GpuMetricStatRow[]; + data: GpuMetricSampleRow[]; +} + +export interface GpuMetricsRunPayload { + workflowRun: { + id: number; + githubRunId: number; + runAttempt: number; + name: string; + date: string; + htmlUrl: string | null; + headBranch: string | null; + headSha: string | null; + conclusion: string | null; + status: string | null; + createdAt: string | null; + }; + series: GpuMetricSeries[]; +} + +interface RawSeriesRow { + id: number | string; + workflow_run_id: number | string; + artifact_name: string; + config_key: string; + file_name: string; + vendor: string; + sample_interval_s: number | null; + sample_count: number; + gpu_count: number; + started_at: string | Date; + ended_at: string | Date; + sidecars: Record | string; + benchmark_result_ids: (number | string)[] | null; +} + +interface RawStatRow { + series_id: number | string; + gpu_index: number; + metric: string; + sample_count: number; + min_value: number; + max_value: number; + mean_value: number; + median_value: number; + p95_value: number; + p99_value: number; + stddev_value: number; +} + +interface RawSampleRow { + series_id: number | string; + gpu_index: number; + sampled_at: string | Date; + power_w: number | null; + temperature_c: number | null; + sm_clock_mhz: number | null; + mem_clock_mhz: number | null; + gpu_util_pct: number | null; + mem_util_pct: number | null; + edge_temp_c: number | null; + mem_temp_c: number | null; + gfx_voltage_mv: number | null; + soc_voltage_mv: number | null; + mem_voltage_mv: number | null; + fclk_mhz: number | null; + socclk_mhz: number | null; + mm_activity_pct: number | null; +} + +const isoString = (value: string | Date): string => + value instanceof Date ? value.toISOString() : new Date(value).toISOString(); + +const optional = (value: number | null): number | undefined => (value === null ? undefined : value); + +function toSampleRow(raw: RawSampleRow): GpuMetricSampleRow { + return { + timestamp: isoString(raw.sampled_at), + index: Number(raw.gpu_index), + power: raw.power_w ?? 0, + temperature: raw.temperature_c ?? 0, + smClock: raw.sm_clock_mhz ?? 0, + memClock: raw.mem_clock_mhz ?? 0, + gpuUtil: raw.gpu_util_pct ?? 0, + memUtil: raw.mem_util_pct ?? 0, + edgeTemp: optional(raw.edge_temp_c), + memTemp: optional(raw.mem_temp_c), + gfxVoltage: optional(raw.gfx_voltage_mv), + socVoltage: optional(raw.soc_voltage_mv), + memVoltage: optional(raw.mem_voltage_mv), + fclk: optional(raw.fclk_mhz), + socClk: optional(raw.socclk_mhz), + mmActivity: optional(raw.mm_activity_pct), + }; +} + +function toStatRow(raw: RawStatRow): GpuMetricStatRow { + return { + gpuIndex: Number(raw.gpu_index), + metric: raw.metric, + count: Number(raw.sample_count), + min: raw.min_value, + max: raw.max_value, + mean: raw.mean_value, + median: raw.median_value, + p95: raw.p95_value, + p99: raw.p99_value, + stddev: raw.stddev_value, + }; +} + +async function loadSeriesDetails( + sql: DbClient, + seriesRows: readonly RawSeriesRow[], + includeSamples: boolean, +): Promise { + if (seriesRows.length === 0) return []; + const ids = seriesRows.map((row) => Number(row.id)); + + const statRows = (await sql` + select series_id, gpu_index, metric, sample_count, min_value, max_value, mean_value, + median_value, p95_value, p99_value, stddev_value + from gpu_metric_gpu_stats + where series_id = any(${ids}::bigint[]) + order by series_id, gpu_index, metric + `) as unknown as RawStatRow[]; + + const sampleRows = includeSamples + ? ((await sql` + select series_id, gpu_index, sampled_at, power_w, temperature_c, sm_clock_mhz, + mem_clock_mhz, gpu_util_pct, mem_util_pct, edge_temp_c, mem_temp_c, + gfx_voltage_mv, soc_voltage_mv, mem_voltage_mv, fclk_mhz, socclk_mhz, mm_activity_pct + from gpu_metric_samples + where series_id = any(${ids}::bigint[]) + order by series_id, sampled_at, gpu_index + `) as unknown as RawSampleRow[]) + : []; + + const statsBySeries = new Map(); + for (const raw of statRows) { + const key = Number(raw.series_id); + const bucket = statsBySeries.get(key); + if (bucket) bucket.push(toStatRow(raw)); + else statsBySeries.set(key, [toStatRow(raw)]); + } + const samplesBySeries = new Map(); + for (const raw of sampleRows) { + const key = Number(raw.series_id); + const bucket = samplesBySeries.get(key); + if (bucket) bucket.push(toSampleRow(raw)); + else samplesBySeries.set(key, [toSampleRow(raw)]); + } + + return seriesRows.map((row) => { + const id = Number(row.id); + return { + id, + artifactName: row.artifact_name, + configKey: row.config_key, + fileName: row.file_name, + vendor: row.vendor, + sampleIntervalS: row.sample_interval_s, + sampleCount: Number(row.sample_count), + gpuCount: Number(row.gpu_count), + startedAt: isoString(row.started_at), + endedAt: isoString(row.ended_at), + sidecars: + typeof row.sidecars === 'string' + ? (JSON.parse(row.sidecars) as Record) + : row.sidecars, + benchmarkResultIds: (row.benchmark_result_ids ?? []).map(Number), + stats: statsBySeries.get(id) ?? [], + data: samplesBySeries.get(id) ?? [], + }; + }); +} + +/** + * Every telemetry series stored for one GitHub Actions run (latest attempt). + * Returns null when the run is unknown or has no stored series, so callers can + * fall back to the live GitHub artifacts for in-flight runs. + */ +export async function getGpuMetricsForRun( + sql: DbClient, + githubRunId: number, + options: { includeSamples?: boolean } = {}, +): Promise { + const runRows = (await sql` + select id, github_run_id, run_attempt, name, date, html_url, head_branch, head_sha, + conclusion, status, created_at + from workflow_runs + where github_run_id = ${githubRunId} + order by run_attempt desc + limit 1 + `) as unknown as { + id: number | string; + github_run_id: number | string; + run_attempt: number; + name: string; + date: string | Date; + html_url: string | null; + head_branch: string | null; + head_sha: string | null; + conclusion: string | null; + status: string | null; + created_at: string | Date | null; + }[]; + const run = runRows[0]; + if (!run) return null; + + const seriesRows = (await sql` + select s.id, s.workflow_run_id, s.artifact_name, s.config_key, s.file_name, s.vendor, + s.sample_interval_s, s.sample_count, s.gpu_count, s.started_at, s.ended_at, s.sidecars, + ( + select array_agg(l.benchmark_result_id order by l.benchmark_result_id) + from benchmark_result_gpu_metrics l where l.series_id = s.id + ) as benchmark_result_ids + from gpu_metric_series s + where s.workflow_run_id = ${Number(run.id)} + order by s.artifact_name, s.file_name + `) as unknown as RawSeriesRow[]; + if (seriesRows.length === 0) return null; + + return { + workflowRun: { + id: Number(run.id), + githubRunId: Number(run.github_run_id), + runAttempt: Number(run.run_attempt), + name: run.name, + date: isoString(run.date).slice(0, 10), + htmlUrl: run.html_url, + headBranch: run.head_branch, + headSha: run.head_sha, + conclusion: run.conclusion, + status: run.status, + createdAt: run.created_at ? isoString(run.created_at) : null, + }, + series: await loadSeriesDetails(sql, seriesRows, options.includeSamples ?? true), + }; +} + +export interface GpuMetricsPointPayload { + benchmarkResultId: number; + series: GpuMetricSeries[]; +} + +/** Series linked to one benchmark point, with samples. Null when none is linked. */ +export async function getGpuMetricsForPoint( + sql: DbClient, + benchmarkResultId: number, +): Promise { + const seriesRows = (await sql` + select s.id, s.workflow_run_id, s.artifact_name, s.config_key, s.file_name, s.vendor, + s.sample_interval_s, s.sample_count, s.gpu_count, s.started_at, s.ended_at, s.sidecars, + ( + select array_agg(l.benchmark_result_id order by l.benchmark_result_id) + from benchmark_result_gpu_metrics l where l.series_id = s.id + ) as benchmark_result_ids + from benchmark_result_gpu_metrics link + join gpu_metric_series s on s.id = link.series_id + where link.benchmark_result_id = ${benchmarkResultId} + order by s.artifact_name, s.file_name + `) as unknown as RawSeriesRow[]; + if (seriesRows.length === 0) return null; + return { + benchmarkResultId, + series: await loadSeriesDetails(sql, seriesRows, true), + }; +} + +/** `benchmark_results.id` → true for each id that has at least one linked series. */ +export async function getGpuMetricsAvailability( + sql: DbClient, + benchmarkResultIds: readonly number[], +): Promise> { + if (benchmarkResultIds.length === 0) return {}; + const rows = (await sql` + select distinct benchmark_result_id + from benchmark_result_gpu_metrics + where benchmark_result_id = any(${[...benchmarkResultIds]}::bigint[]) + `) as unknown as { benchmark_result_id: number | string }[]; + const out: Record = {}; + for (const row of rows) out[Number(row.benchmark_result_id)] = true; + return out; +} diff --git a/packages/db/src/verify-db.ts b/packages/db/src/verify-db.ts index 1f4d51edd..3ab8f805c 100644 --- a/packages/db/src/verify-db.ts +++ b/packages/db/src/verify-db.ts @@ -50,6 +50,9 @@ async function verify(): Promise { TABLE_NAMES.evalResults, TABLE_NAMES.evalSamples, TABLE_NAMES.changelogEntries, + TABLE_NAMES.gpuMetricSeries, + TABLE_NAMES.gpuMetricSamples, + TABLE_NAMES.benchmarkResultGpuMetrics, ]; for (const t of tables) { const [{ n }] = await sql`select count(*)::int as n from ${sql(t)}`; From 54ae365b51acfb7d9dc112be8322b80514992e9a Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 17 Sep 2026 13:52:28 -0500 Subject: [PATCH 002/103] feat(powerx): add rolling-average display mode and chip toggles to telemetry charts Add a Points / Rolling average control to the PowerX telemetry charts on both the explorer page and the per-point PowerX tab, with a 10/30/60/300 s window selector. The average is a centered time-window mean per chip (pure helper in telemetry-smoothing.ts, unit-tested for empty input, single sample, irregular timestamps and inclusive window edges), so it follows elapsed time rather than sample count. Averaged mode draws smooth lines and keeps the sample circles as invisible hover targets so the tooltip and crosshair still work. The per-point PowerX tab gains the same chip legend as the explorer so individual chips can be hidden and restored, scoped to the selected series. The chart's t=0 now comes from the whole series rather than the visible chips so hiding a chip does not shift the time axis. Co-Authored-By: Claude Fable 5.1 --- .../components/gpu-power/GpuPowerChart.tsx | 80 ++++++++++--- .../components/gpu-power/GpuPowerDisplay.tsx | 17 +++ .../gpu-power/TelemetryDisplayControls.tsx | 111 ++++++++++++++++++ .../gpu-power/telemetry-smoothing.test.ts | 70 +++++++++++ .../gpu-power/telemetry-smoothing.ts | 61 ++++++++++ .../agentic-point/power-telemetry-view.tsx | 88 +++++++++++++- 6 files changed, 409 insertions(+), 18 deletions(-) create mode 100644 packages/app/src/components/gpu-power/TelemetryDisplayControls.tsx create mode 100644 packages/app/src/components/gpu-power/telemetry-smoothing.test.ts create mode 100644 packages/app/src/components/gpu-power/telemetry-smoothing.ts diff --git a/packages/app/src/components/gpu-power/GpuPowerChart.tsx b/packages/app/src/components/gpu-power/GpuPowerChart.tsx index e26796b1a..62447c3e9 100644 --- a/packages/app/src/components/gpu-power/GpuPowerChart.tsx +++ b/packages/app/src/components/gpu-power/GpuPowerChart.tsx @@ -5,6 +5,11 @@ import React, { useMemo } from 'react'; import { D3Chart } from '@/lib/d3-chart/D3Chart'; import { useLocale } from '@/lib/use-locale'; +import { + DEFAULT_TELEMETRY_DISPLAY, + rollingTimeAverage, + type TelemetryDisplayState, +} from './telemetry-smoothing'; import { type GpuMetricKey, type GpuMetricRow, @@ -25,6 +30,7 @@ const STRINGS = { power: 'Power', temp: 'Temp', utilization: 'Chip Util', + rollingSuffix: (windowS: number) => `${windowS} s rolling avg`, }, zh: { empty: '暂无可显示的芯片指标数据。', @@ -35,14 +41,18 @@ const STRINGS = { power: '功耗', temp: '温度', utilization: '芯片利用率', + rollingSuffix: (windowS: number) => `${windowS} 秒滚动平均`, }, } as const; interface ParsedPoint { seconds: number; + /** Absolute sample time in ms; smoothing and alignment work in this space. */ + ms: number; value: number; gpuIndex: number; - raw: GpuMetricRow; + /** The raw sample behind this point; null once the value has been averaged. */ + raw: GpuMetricRow | null; } interface GpuMetricsChartProps { @@ -54,6 +64,8 @@ interface GpuMetricsChartProps { caption?: React.ReactNode; /** Max interactive points before LTTB downsampling. Infinity to disable. */ maxPoints?: number; + /** Raw samples vs. time-window rolling average. Defaults to raw samples. */ + display?: TelemetryDisplayState; } function parseTimestamp(raw: string): Date | null { @@ -71,15 +83,16 @@ function buildGroupedData( visibleGpus: Set, metricKey: GpuMetricKey, ): Map { + // t=0 is the first sample of the whole series, not of the visible chips, so + // hiding a chip never shifts the time axis under the remaining lines. let minTime = Infinity; const parsed: { row: GpuMetricRow; ms: number }[] = []; for (const row of data) { - if (!visibleGpus.has(row.index)) continue; const time = parseTimestamp(row.timestamp); if (!time) continue; const ms = time.getTime(); - parsed.push({ row, ms }); if (ms < minTime) minTime = ms; + if (visibleGpus.has(row.index)) parsed.push({ row, ms }); } const groups = new Map(); @@ -87,6 +100,7 @@ function buildGroupedData( if (!groups.has(row.index)) groups.set(row.index, []); groups.get(row.index)!.push({ seconds: (ms - minTime) / 1000, + ms, value: row[metricKey] ?? 0, gpuIndex: row.index, raw: row, @@ -98,7 +112,14 @@ function buildGroupedData( return groups; } -const GPU_COLORS = d3.schemeTableau10; +/** Replace each point's value with its centered time-window mean. */ +function smoothSeries(points: ParsedPoint[], windowS: number): ParsedPoint[] { + const averaged = rollingTimeAverage(points, windowS * 1000); + return points.map((point, i) => ({ ...point, value: averaged[i]!.value, raw: null })); +} + +/** Per-chip palette shared with legends that toggle chips on and off. */ +export const GPU_COLORS = d3.schemeTableau10; const CHART_ID = 'gpu-metrics-line'; const MARGIN = { top: 24, right: 20, bottom: 60, left: 60 }; @@ -111,16 +132,27 @@ const GpuMetricsChart = React.memo( legendElement, caption, maxPoints, + display = DEFAULT_TELEMETRY_DISPLAY, }: GpuMetricsChartProps) => { const locale = useLocale(); const t = STRINGS[locale]; const metricConfig = ALL_METRIC_OPTIONS.find((m) => m.key === metricKey)!; + const rolling = display.mode === 'rolling'; - const groupedData = useMemo( + const rawGroups = useMemo( () => buildGroupedData(data, visibleGpus, metricKey), [data, visibleGpus, metricKey], ); + const groupedData = useMemo(() => { + if (!rolling) return rawGroups; + const smoothed = new Map(); + for (const [gpuIndex, points] of rawGroups) { + smoothed.set(gpuIndex, smoothSeries(points, display.windowS)); + } + return smoothed; + }, [rawGroups, rolling, display.windowS]); + const allPoints = useMemo(() => { const pts: ParsedPoint[] = []; for (const points of groupedData.values()) pts.push(...points); @@ -216,7 +248,7 @@ const GpuMetricsChart = React.memo( lines: lineData, config: { getColor: (key) => GPU_COLORS[parseInt(key, 10) % GPU_COLORS.length], - strokeWidth: 1.5, + strokeWidth: rolling ? 1.75 : 1.5, curve: d3.curveMonotoneX, }, }, @@ -230,8 +262,11 @@ const GpuMetricsChart = React.memo( getCy: () => 0, getX: (d) => d.seconds, getY: (d) => d.value, - getColor: (d) => GPU_COLORS[d.gpuIndex % GPU_COLORS.length], - getRadius: () => 2, + // Averaged mode draws lines only; the circles stay as invisible + // hover targets so the tooltip and crosshair keep working. + getColor: (d) => + rolling ? 'transparent' : GPU_COLORS[d.gpuIndex % GPU_COLORS.length], + getRadius: () => (rolling ? 3 : 2), maxPoints, }, }, @@ -246,23 +281,36 @@ const GpuMetricsChart = React.memo( rulerType: 'crosshair', content: (d: ParsedPoint, isPinned: boolean) => { const color = GPU_COLORS[d.gpuIndex % GPU_COLORS.length]; + const sep = locale === 'zh' ? ':' : ':'; return `
${isPinned ? `
${t.dismiss}
` : ''}
${t.chip} ${d.gpuIndex}
${d.seconds.toFixed(1)}${locale === 'zh' ? ' 秒' : 's'}
-
${getGpuMetricLabel(metricConfig, locale)}${locale === 'zh' ? ':' : ':'} ${d.value.toFixed(1)} ${metricConfig.unit}
-
${t.power}${locale === 'zh' ? ':' : ':'} ${d.raw.power.toFixed(1)} W
-
${t.temp}${locale === 'zh' ? ':' : ':'} ${d.raw.temperature}\u00B0C
-
${t.utilization}${locale === 'zh' ? ':' : ':'} ${d.raw.gpuUtil}%
+
${getGpuMetricLabel(metricConfig, locale)}${sep} ${d.value.toFixed(1)} ${metricConfig.unit}
+ ${rolling ? `
${t.rollingSuffix(display.windowS)}
` : ''} + ${ + d.raw + ? `
${t.power}${sep} ${d.raw.power.toFixed(1)} W
+
${t.temp}${sep} ${d.raw.temperature}\u00B0C
+
${t.utilization}${sep} ${d.raw.gpuUtil}%
` + : '' + }
`; }, getRulerX: (d, xScale) => (xScale as d3.ScaleLinear)(d.seconds), getRulerY: (d, yScale) => yScale(d.value), - onHoverStart: (sel) => { - sel.attr('r', 5).attr('stroke', 'white').attr('stroke-width', 1); + onHoverStart: (sel, d) => { + sel + .attr('r', 5) + .attr('fill', GPU_COLORS[d.gpuIndex % GPU_COLORS.length]) + .attr('stroke', 'white') + .attr('stroke-width', 1); }, - onHoverEnd: (sel) => { - sel.attr('r', 2).attr('stroke', 'none'); + onHoverEnd: (sel, d) => { + sel + .attr('r', rolling ? 3 : 2) + .attr('fill', rolling ? 'transparent' : GPU_COLORS[d.gpuIndex % GPU_COLORS.length]) + .attr('stroke', 'none'); }, attachToLayer: 2, }} diff --git a/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx b/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx index f0949589c..544795ed7 100644 --- a/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx +++ b/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx @@ -33,6 +33,8 @@ import { useClientSearchParams } from '@/hooks/useClientSearch'; import GpuCorrelationChart from './GpuCorrelationChart'; import GpuMetricsChart from './GpuPowerChart'; import GpuStatsTable from './GpuStatsTable'; +import { DEFAULT_TELEMETRY_DISPLAY, type TelemetryDisplayState } from './telemetry-smoothing'; +import { TelemetryDisplayControls } from './TelemetryDisplayControls'; import { type GpuMetricKey, type GpuPowerApiResponse, @@ -81,6 +83,7 @@ const STRINGS = { chip: 'Chip', chartToolbar: 'Chart controls', correlationAxes: 'Correlation axes', + displayControls: 'Line options', }, zh: { heading: 'PowerX', @@ -119,6 +122,7 @@ const STRINGS = { chip: '芯片', chartToolbar: '图表控制', correlationAxes: '相关性坐标轴', + displayControls: '曲线选项', }, } as const; @@ -237,6 +241,7 @@ export default function GpuMetricsDisplay() { const [chartView, setChartView] = useState('chart'); const [corrXMetric, setCorrXMetric] = useState('power'); const [corrYMetric, setCorrYMetric] = useState('temperature'); + const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); const viewOptions = useMemo[]>( () => [ { @@ -627,6 +632,17 @@ export default function GpuMetricsDisplay() { )} + {chartView === 'chart' && ( + + + + )} + {chartView === 'chart' && (

diff --git a/packages/app/src/components/gpu-power/TelemetryDisplayControls.tsx b/packages/app/src/components/gpu-power/TelemetryDisplayControls.tsx new file mode 100644 index 000000000..82fedd75b --- /dev/null +++ b/packages/app/src/components/gpu-power/TelemetryDisplayControls.tsx @@ -0,0 +1,111 @@ +'use client'; + +import { Label } from '@/components/ui/label'; +import { SegmentedToggle, type SegmentedToggleOption } from '@/components/ui/segmented-toggle'; +import { + Select, + SelectContent, + SelectItem, + SelectTrigger, + SelectValue, +} from '@/components/ui/select'; +import { track } from '@/lib/analytics'; +import { useLocale } from '@/lib/use-locale'; + +import { + SMOOTHING_WINDOWS_S, + type SmoothingWindowS, + type TelemetryDisplayMode, + type TelemetryDisplayState, +} from './telemetry-smoothing'; + +const STRINGS = { + en: { + display: 'Display', + points: 'Points', + rolling: 'Rolling average', + window: 'Window', + windowOption: (s: number) => `${s} s`, + }, + zh: { + display: '显示方式', + points: '数据点', + rolling: '滚动平均', + window: '窗口', + windowOption: (s: number) => `${s} 秒`, + }, +} as const; + +interface Props { + value: TelemetryDisplayState; + onChange: (next: TelemetryDisplayState) => void; + /** Analytics event prefix, e.g. `gpu_metrics` → `gpu_metrics_display_mode_changed`. */ + analyticsPrefix: string; + /** Prefix for control ids so two charts on one page do not collide. */ + idPrefix: string; + className?: string; +} + +/** + * Display-mode controls shared by the PowerX explorer and the per-point PowerX + * tab: raw samples vs. a time-window rolling average. + */ +export function TelemetryDisplayControls({ + value, + onChange, + analyticsPrefix, + idPrefix, + className, +}: Props) { + const locale = useLocale(); + const t = STRINGS[locale]; + + const modeOptions: SegmentedToggleOption[] = [ + { value: 'points', label: t.points, testId: `${idPrefix}-mode-points` }, + { value: 'rolling', label: t.rolling, testId: `${idPrefix}-mode-rolling` }, + ]; + + return ( +
+
+
+ + { + track(`${analyticsPrefix}_display_mode_changed`, { mode }); + onChange({ ...value, mode }); + }} + /> +
+ {value.mode === 'rolling' && ( +
+ + +
+ )} +
+
+ ); +} diff --git a/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts b/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts new file mode 100644 index 000000000..f33793617 --- /dev/null +++ b/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts @@ -0,0 +1,70 @@ +import { describe, expect, it } from 'vitest'; + +import { rollingTimeAverage, type TimedSample } from './telemetry-smoothing'; + +const T0 = Date.parse('2026-09-15T21:36:14.107Z'); + +/** One sample per second starting at T0, values from `values`. */ +function everySecond(values: number[], offsetMs = 0): TimedSample[] { + return values.map((value, i) => ({ ms: T0 + offsetMs + i * 1000, value })); +} + +describe('rollingTimeAverage', () => { + it('returns an empty array for empty input', () => { + expect(rollingTimeAverage([], 30_000)).toEqual([]); + }); + + it('returns the sample unchanged for a single sample', () => { + expect(rollingTimeAverage([{ ms: T0, value: 812.4 }], 60_000)).toEqual([ + { ms: T0, value: 812.4 }, + ]); + }); + + it('copies the input when the window is zero or negative', () => { + const input = everySecond([1, 2, 3]); + expect(rollingTimeAverage(input, 0)).toEqual(input); + expect(rollingTimeAverage(input, -5)).toEqual(input); + }); + + it('averages a centered window and shrinks it at the series edges', () => { + // 1 s cadence, 2 s window => each sample sees itself and its immediate neighbours. + const out = rollingTimeAverage(everySecond([100, 200, 600, 200, 100]), 2000); + expect(out.map((p) => p.value)).toEqual([150, 300, 1000 / 3, 300, 150]); + expect(out.map((p) => p.ms)).toEqual(everySecond([0, 0, 0, 0, 0]).map((p) => p.ms)); + }); + + it('treats both window edges as inclusive', () => { + // 4 s window => samples exactly 2 s away are included, 3 s away are not. + const out = rollingTimeAverage(everySecond([0, 0, 90, 0, 0, 0]), 4000); + // index 2 sees indices 0..4 (five samples) => 18; index 5 sees 3..5 => 0. + expect(out[2]!.value).toBeCloseTo(18); + expect(out[5]!.value).toBe(0); + // index 0 sees 0..2 => 30. + expect(out[0]!.value).toBeCloseTo(30); + }); + + it('uses elapsed time, not sample count, for irregular timestamps', () => { + const input: TimedSample[] = [ + { ms: T0, value: 100 }, + { ms: T0 + 500, value: 300 }, // burst of close samples + { ms: T0 + 700, value: 300 }, + { ms: T0 + 60_000, value: 900 }, // one minute gap: outside any 10 s window + ]; + const out = rollingTimeAverage(input, 10_000); + expect(out[0]!.value).toBeCloseTo((100 + 300 + 300) / 3); + expect(out[3]!.value).toBe(900); + }); + + it('smooths a step change into a ramp whose plateaus keep the original level', () => { + const values = [ + ...Array.from({ length: 30 }, () => 250), + ...Array.from({ length: 30 }, () => 950), + ]; + const out = rollingTimeAverage(everySecond(values), 10_000); + expect(out[5]!.value).toBe(250); + expect(out[54]!.value).toBe(950); + const ramp = out.slice(25, 35).map((p) => p.value); + for (let i = 1; i < ramp.length; i += 1) expect(ramp[i]!).toBeGreaterThanOrEqual(ramp[i - 1]!); + expect(ramp.at(-1)!).toBeGreaterThan(ramp[0]!); + }); +}); diff --git a/packages/app/src/components/gpu-power/telemetry-smoothing.ts b/packages/app/src/components/gpu-power/telemetry-smoothing.ts new file mode 100644 index 000000000..c4ee00e82 --- /dev/null +++ b/packages/app/src/components/gpu-power/telemetry-smoothing.ts @@ -0,0 +1,61 @@ +/** + * Pure time-series helpers for the PowerX telemetry charts. They operate on + * absolute millisecond timestamps so irregular sampling (dropped rows, a few + * ms of skew between chips) is handled by time, not by sample index. + */ + +export interface TimedSample { + /** Absolute timestamp in milliseconds since the Unix epoch. */ + ms: number; + value: number; +} + +/** Rolling-average window choices offered by the chart controls, in seconds. */ +export const SMOOTHING_WINDOWS_S = [10, 30, 60, 300] as const; +export type SmoothingWindowS = (typeof SMOOTHING_WINDOWS_S)[number]; + +export type TelemetryDisplayMode = 'points' | 'rolling'; + +export interface TelemetryDisplayState { + mode: TelemetryDisplayMode; + windowS: SmoothingWindowS; +} + +export const DEFAULT_TELEMETRY_DISPLAY: TelemetryDisplayState = { + mode: 'points', + windowS: 30, +}; + +/** + * Centered time-window mean. Each output sample averages every input sample + * whose timestamp lies within `windowMs / 2` (inclusive) of its own, so the + * smoothed line stays time-aligned with the raw one instead of lagging by half + * a window as a trailing average would. Window edges are inclusive on both + * sides; samples near the start or end of the series average over the shorter + * one-sided neighbourhood that exists. + * + * `samples` must be sorted by `ms` ascending. O(n) via prefix sums. + */ +export function rollingTimeAverage( + samples: readonly TimedSample[], + windowMs: number, +): TimedSample[] { + if (samples.length === 0) return []; + if (!(windowMs > 0)) return samples.map(({ ms, value }) => ({ ms, value })); + const half = windowMs / 2; + const n = samples.length; + const prefix = new Float64Array(n + 1); + for (let i = 0; i < n; i += 1) prefix[i + 1] = prefix[i]! + samples[i]!.value; + + const out: TimedSample[] = Array.from({ length: n }); + let lo = 0; + let hi = 0; + for (let i = 0; i < n; i += 1) { + const center = samples[i]!.ms; + while (samples[lo]!.ms < center - half) lo += 1; + while (hi < n && samples[hi]!.ms <= center + half) hi += 1; + const count = hi - lo; + out[i] = { ms: center, value: (prefix[hi]! - prefix[lo]!) / count }; + } + return out; +} diff --git a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx index 8337eedb7..f50c7d568 100644 --- a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx +++ b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx @@ -2,8 +2,13 @@ import { useMemo, useState } from 'react'; -import GpuMetricsChart from '@/components/gpu-power/GpuPowerChart'; +import GpuMetricsChart, { GPU_COLORS } from '@/components/gpu-power/GpuPowerChart'; import GpuStatsTable from '@/components/gpu-power/GpuStatsTable'; +import { TelemetryDisplayControls } from '@/components/gpu-power/TelemetryDisplayControls'; +import { + DEFAULT_TELEMETRY_DISPLAY, + type TelemetryDisplayState, +} from '@/components/gpu-power/telemetry-smoothing'; import { type GpuMetricKey, type GpuMetricRow, @@ -12,6 +17,7 @@ import { getGpuMetricLabel, } from '@/components/gpu-power/types'; import { Card } from '@/components/ui/card'; +import ChartLegend from '@/components/ui/chart-legend'; import { Label } from '@/components/ui/label'; import { RetryableQueryError } from '@/components/ui/retryable-query-error'; import { @@ -43,6 +49,7 @@ const STRINGS = { perGpuStats: 'Per-chip statistics', chip: 'Chip', secondsUnit: 's', + resetFilter: 'Show all chips', }, zh: { loading: '正在加载 PowerX 遥测数据……', @@ -61,6 +68,7 @@ const STRINGS = { perGpuStats: '单芯片统计信息', chip: '芯片', secondsUnit: '秒', + resetFilter: '显示全部芯片', }, } as const; @@ -101,7 +109,39 @@ export function PowerTelemetryView({ id, enabled }: Props) { ? metricSelection : 'power'; const metricConfig = ALL_METRIC_OPTIONS.find((m) => m.key === metricKey)!; - const visibleGpus = useMemo(() => new Set(data.map((row) => row.index)), [data]); + const allGpuIndices = useMemo( + () => [...new Set(data.map((row) => row.index))].toSorted((a, b) => a - b), + [data], + ); + // Hidden chips are scoped to the series they were hidden on so switching + // series never carries over a stale filter. + const [hiddenSelection, setHiddenSelection] = useState<{ + seriesId: number; + hidden: number[]; + } | null>(null); + const hiddenGpus = useMemo( + () => + new Set( + hiddenSelection && hiddenSelection.seriesId === selectedSeries?.id + ? hiddenSelection.hidden + : [], + ), + [hiddenSelection, selectedSeries?.id], + ); + const visibleGpus = useMemo( + () => new Set(allGpuIndices.filter((gpuIndex) => !hiddenGpus.has(gpuIndex))), + [allGpuIndices, hiddenGpus], + ); + const toggleGpu = (gpuIndex: number) => { + if (!selectedSeries) return; + track('inference_agentic_power_gpu_toggled', { id, gpuIndex }); + const next = new Set(hiddenGpus); + if (next.has(gpuIndex)) next.delete(gpuIndex); + else next.add(gpuIndex); + setHiddenSelection({ seriesId: selectedSeries.id, hidden: [...next] }); + }; + const [isLegendExpanded, setIsLegendExpanded] = useState(true); + const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); if (!enabled) return null; @@ -229,6 +269,13 @@ export function PowerTelemetryView({ id, enabled }: Props) { + @@ -238,6 +285,43 @@ export function PowerTelemetryView({ id, enabled }: Props) { metricKey={metricKey} artifactName={selectedSeries.artifactName} maxPoints={2000} + display={display} + legendElement={ + ({ + name: `${t.chip} ${gpuIndex}`, + hw: String(gpuIndex), + label: `${t.chip} ${gpuIndex}`, + color: GPU_COLORS[gpuIndex % GPU_COLORS.length], + isActive: visibleGpus.has(gpuIndex), + onClick: () => toggleGpu(gpuIndex), + }))} + onItemRemove={(hw) => { + const gpuIndex = Number(hw); + if (visibleGpus.has(gpuIndex)) toggleGpu(gpuIndex); + }} + isLegendExpanded={isLegendExpanded} + onExpandedChange={(expanded) => { + setIsLegendExpanded(expanded); + track('inference_agentic_power_legend_expanded', { id, expanded }); + }} + actions={ + hiddenGpus.size === 0 + ? [] + : [ + { + id: 'power-telemetry-show-all-chips', + label: t.resetFilter, + onClick: () => { + track('inference_agentic_power_gpu_reset_filter', { id }); + setHiddenSelection(null); + }, + }, + ] + } + /> + } caption={ {getGpuMetricLabel(metricConfig, locale)} · {t.sharedNote} From 5422a82cb0eb4a0ce03de073c2fd9096a6f80b2d Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 17 Sep 2026 13:57:13 -0500 Subject: [PATCH 003/103] feat(powerx): add mean-across-chips line to telemetry charts Add a Lines control (Per chip / Mean of chips / Both) next to the display-mode toggle on both PowerX surfaces. The mean line averages the currently visible chips at each timestamp of the longest series, aligning the other chips by nearest sample within one estimated poll interval (median gap), so the few ms of skew between nvidia-smi rows and the occasional dropped row do not interpolate or misalign. It is drawn in the foreground color at a heavier stroke, labelled in a key row under the caption, and its tooltip reports how many chips contributed. The rolling-average mode smooths the mean line too. The shared line layer gains an optional per-series getStrokeWidth so one keyed join can hold both the chip lines and the heavier mean line. Co-Authored-By: Claude Fable 5.1 --- .../components/gpu-power/GpuPowerChart.tsx | 105 +++++++++++++++--- .../gpu-power/TelemetryDisplayControls.tsx | 30 ++++- .../gpu-power/telemetry-smoothing.test.ts | 94 +++++++++++++++- .../gpu-power/telemetry-smoothing.ts | 72 ++++++++++++ .../app/src/lib/d3-chart/layers/lines.test.ts | 23 ++++ packages/app/src/lib/d3-chart/layers/lines.ts | 4 +- 6 files changed, 309 insertions(+), 19 deletions(-) diff --git a/packages/app/src/components/gpu-power/GpuPowerChart.tsx b/packages/app/src/components/gpu-power/GpuPowerChart.tsx index 62447c3e9..17a9f4790 100644 --- a/packages/app/src/components/gpu-power/GpuPowerChart.tsx +++ b/packages/app/src/components/gpu-power/GpuPowerChart.tsx @@ -7,6 +7,8 @@ import { D3Chart } from '@/lib/d3-chart/D3Chart'; import { useLocale } from '@/lib/use-locale'; import { DEFAULT_TELEMETRY_DISPLAY, + estimateSampleIntervalMs, + meanAcrossSeries, rollingTimeAverage, type TelemetryDisplayState, } from './telemetry-smoothing'; @@ -31,6 +33,8 @@ const STRINGS = { temp: 'Temp', utilization: 'Chip Util', rollingSuffix: (windowS: number) => `${windowS} s rolling avg`, + meanChips: 'Mean of visible chips', + meanCount: (n: number) => `${n} chips`, }, zh: { empty: '暂无可显示的芯片指标数据。', @@ -42,6 +46,8 @@ const STRINGS = { temp: '温度', utilization: '芯片利用率', rollingSuffix: (windowS: number) => `${windowS} 秒滚动平均`, + meanChips: '可见芯片均值', + meanCount: (n: number) => `${n} 个芯片`, }, } as const; @@ -53,6 +59,8 @@ interface ParsedPoint { gpuIndex: number; /** The raw sample behind this point; null once the value has been averaged. */ raw: GpuMetricRow | null; + /** For the mean line: how many chips contributed at this timestamp. */ + count?: number; } interface GpuMetricsChartProps { @@ -82,7 +90,7 @@ function buildGroupedData( data: GpuMetricRow[], visibleGpus: Set, metricKey: GpuMetricKey, -): Map { +): { t0Ms: number; groups: Map } { // t=0 is the first sample of the whole series, not of the visible chips, so // hiding a chip never shifts the time axis under the remaining lines. let minTime = Infinity; @@ -109,7 +117,23 @@ function buildGroupedData( for (const points of groups.values()) { points.sort((a, b) => a.seconds - b.seconds); } - return groups; + return { t0Ms: minTime, groups }; +} + +/** Mean across the visible chips, aligned by nearest sample within one poll interval. */ +function buildMeanSeries(groups: Map, t0Ms: number): ParsedPoint[] { + const arrays = [...groups.values()]; + if (arrays.length === 0) return []; + const longest = arrays.reduce((a, b) => (b.length > a.length ? b : a)); + const mean = meanAcrossSeries(arrays, estimateSampleIntervalMs(longest)); + return mean.map((p) => ({ + seconds: (p.ms - t0Ms) / 1000, + ms: p.ms, + value: p.value, + gpuIndex: MEAN_INDEX, + raw: null, + count: p.count, + })); } /** Replace each point's value with its centered time-window mean. */ @@ -120,6 +144,18 @@ function smoothSeries(points: ParsedPoint[], windowS: number): ParsedPoint[] { /** Per-chip palette shared with legends that toggle chips on and off. */ export const GPU_COLORS = d3.schemeTableau10; +/** Pseudo chip index and line key for the mean across visible chips. */ +const MEAN_INDEX = -1; +const MEAN_KEY = 'mean'; +const MEAN_COLOR = 'var(--foreground)'; + +function lineKey(gpuIndex: number): string { + return gpuIndex === MEAN_INDEX ? MEAN_KEY : String(gpuIndex); +} + +function colorFor(gpuIndex: number): string { + return gpuIndex === MEAN_INDEX ? MEAN_COLOR : GPU_COLORS[gpuIndex % GPU_COLORS.length]!; +} const CHART_ID = 'gpu-metrics-line'; const MARGIN = { top: 24, right: 20, bottom: 60, left: 60 }; @@ -138,20 +174,30 @@ const GpuMetricsChart = React.memo( const t = STRINGS[locale]; const metricConfig = ALL_METRIC_OPTIONS.find((m) => m.key === metricKey)!; const rolling = display.mode === 'rolling'; + const showChips = display.series !== 'mean'; + const showMean = display.series !== 'chips'; - const rawGroups = useMemo( + const { t0Ms, groups: rawGroups } = useMemo( () => buildGroupedData(data, visibleGpus, metricKey), [data, visibleGpus, metricKey], ); + // Displayed series keyed by chip index (MEAN_INDEX for the mean line): + // per-chip and/or mean, then optionally smoothed. const groupedData = useMemo(() => { - if (!rolling) return rawGroups; + const series = new Map(); + if (showChips) for (const [gpuIndex, points] of rawGroups) series.set(gpuIndex, points); + if (showMean) { + const mean = buildMeanSeries(rawGroups, t0Ms); + if (mean.length > 0) series.set(MEAN_INDEX, mean); + } + if (!rolling) return series; const smoothed = new Map(); - for (const [gpuIndex, points] of rawGroups) { + for (const [gpuIndex, points] of series) { smoothed.set(gpuIndex, smoothSeries(points, display.windowS)); } return smoothed; - }, [rawGroups, rolling, display.windowS]); + }, [rawGroups, t0Ms, showChips, showMean, rolling, display.windowS]); const allPoints = useMemo(() => { const pts: ParsedPoint[] = []; @@ -163,11 +209,35 @@ const GpuMetricsChart = React.memo( const lineData = useMemo(() => { const result: Record = {}; for (const [gpuIndex, points] of groupedData) { - result[String(gpuIndex)] = points.map((p) => ({ x: p.seconds, y: p.value })); + result[lineKey(gpuIndex)] = points.map((p) => ({ x: p.seconds, y: p.value })); } return result; }, [groupedData]); + const hasMeanLine = groupedData.has(MEAN_INDEX); + const keyRow = hasMeanLine ? ( +
+ + + {t.meanChips} + +
+ ) : null; + const resolvedCaption = + caption || keyRow ? ( + <> + {caption} + {keyRow} + + ) : undefined; + // Scale domains const xDomain = useMemo(() => { if (allPoints.length === 0) return [0, 100] as [number, number]; @@ -247,8 +317,8 @@ const GpuMetricsChart = React.memo( key: 'gpu-lines', lines: lineData, config: { - getColor: (key) => GPU_COLORS[parseInt(key, 10) % GPU_COLORS.length], - strokeWidth: rolling ? 1.75 : 1.5, + getColor: (key) => (key === MEAN_KEY ? MEAN_COLOR : colorFor(parseInt(key, 10))), + getStrokeWidth: (key) => (key === MEAN_KEY ? 2.5 : rolling ? 1.75 : 1.5), curve: d3.curveMonotoneX, }, }, @@ -264,8 +334,7 @@ const GpuMetricsChart = React.memo( getY: (d) => d.value, // Averaged mode draws lines only; the circles stay as invisible // hover targets so the tooltip and crosshair keep working. - getColor: (d) => - rolling ? 'transparent' : GPU_COLORS[d.gpuIndex % GPU_COLORS.length], + getColor: (d) => (rolling ? 'transparent' : colorFor(d.gpuIndex)), getRadius: () => (rolling ? 3 : 2), maxPoints, }, @@ -280,11 +349,15 @@ const GpuMetricsChart = React.memo( tooltip={{ rulerType: 'crosshair', content: (d: ParsedPoint, isPinned: boolean) => { - const color = GPU_COLORS[d.gpuIndex % GPU_COLORS.length]; + const color = colorFor(d.gpuIndex); const sep = locale === 'zh' ? ':' : ':'; + const title = + d.gpuIndex === MEAN_INDEX + ? `${t.meanChips}${d.count ? ` · ${t.meanCount(d.count)}` : ''}` + : `${t.chip} ${d.gpuIndex}`; return `
${isPinned ? `
${t.dismiss}
` : ''} -
${t.chip} ${d.gpuIndex}
+
${title}
${d.seconds.toFixed(1)}${locale === 'zh' ? ' 秒' : 's'}
${getGpuMetricLabel(metricConfig, locale)}${sep} ${d.value.toFixed(1)} ${metricConfig.unit}
${rolling ? `
${t.rollingSuffix(display.windowS)}
` : ''} @@ -302,20 +375,20 @@ const GpuMetricsChart = React.memo( onHoverStart: (sel, d) => { sel .attr('r', 5) - .attr('fill', GPU_COLORS[d.gpuIndex % GPU_COLORS.length]) + .attr('fill', colorFor(d.gpuIndex)) .attr('stroke', 'white') .attr('stroke-width', 1); }, onHoverEnd: (sel, d) => { sel .attr('r', rolling ? 3 : 2) - .attr('fill', rolling ? 'transparent' : GPU_COLORS[d.gpuIndex % GPU_COLORS.length]) + .attr('fill', rolling ? 'transparent' : colorFor(d.gpuIndex)) .attr('stroke', 'none'); }, attachToLayer: 2, }} legendElement={legendElement} - caption={caption} + caption={resolvedCaption} /> ); }, diff --git a/packages/app/src/components/gpu-power/TelemetryDisplayControls.tsx b/packages/app/src/components/gpu-power/TelemetryDisplayControls.tsx index 82fedd75b..816de67e3 100644 --- a/packages/app/src/components/gpu-power/TelemetryDisplayControls.tsx +++ b/packages/app/src/components/gpu-power/TelemetryDisplayControls.tsx @@ -17,6 +17,7 @@ import { type SmoothingWindowS, type TelemetryDisplayMode, type TelemetryDisplayState, + type TelemetrySeriesMode, } from './telemetry-smoothing'; const STRINGS = { @@ -26,6 +27,10 @@ const STRINGS = { rolling: 'Rolling average', window: 'Window', windowOption: (s: number) => `${s} s`, + series: 'Lines', + chips: 'Per chip', + mean: 'Mean of chips', + both: 'Both', }, zh: { display: '显示方式', @@ -33,6 +38,10 @@ const STRINGS = { rolling: '滚动平均', window: '窗口', windowOption: (s: number) => `${s} 秒`, + series: '曲线', + chips: '单芯片', + mean: '芯片均值', + both: '两者', }, } as const; @@ -48,7 +57,8 @@ interface Props { /** * Display-mode controls shared by the PowerX explorer and the per-point PowerX - * tab: raw samples vs. a time-window rolling average. + * tab: raw samples vs. a time-window rolling average, and per-chip lines vs. + * the mean across the visible chips. */ export function TelemetryDisplayControls({ value, @@ -64,6 +74,11 @@ export function TelemetryDisplayControls({ { value: 'points', label: t.points, testId: `${idPrefix}-mode-points` }, { value: 'rolling', label: t.rolling, testId: `${idPrefix}-mode-rolling` }, ]; + const seriesOptions: SegmentedToggleOption[] = [ + { value: 'chips', label: t.chips, testId: `${idPrefix}-series-chips` }, + { value: 'mean', label: t.mean, testId: `${idPrefix}-series-mean` }, + { value: 'both', label: t.both, testId: `${idPrefix}-series-both` }, + ]; return (
@@ -105,6 +120,19 @@ export function TelemetryDisplayControls({
)} +
+ + { + track(`${analyticsPrefix}_series_mode_changed`, { series }); + onChange({ ...value, series }); + }} + /> +
); diff --git a/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts b/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts index f33793617..a9d41298e 100644 --- a/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts +++ b/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts @@ -1,6 +1,11 @@ import { describe, expect, it } from 'vitest'; -import { rollingTimeAverage, type TimedSample } from './telemetry-smoothing'; +import { + estimateSampleIntervalMs, + meanAcrossSeries, + rollingTimeAverage, + type TimedSample, +} from './telemetry-smoothing'; const T0 = Date.parse('2026-09-15T21:36:14.107Z'); @@ -68,3 +73,90 @@ describe('rollingTimeAverage', () => { expect(ramp.at(-1)!).toBeGreaterThan(ramp[0]!); }); }); + +describe('estimateSampleIntervalMs', () => { + it('falls back when there are fewer than two distinct timestamps', () => { + expect(estimateSampleIntervalMs([], 1000)).toBe(1000); + expect(estimateSampleIntervalMs([{ ms: T0, value: 1 }], 750)).toBe(750); + expect( + estimateSampleIntervalMs( + [ + { ms: T0, value: 1 }, + { ms: T0, value: 2 }, + ], + 250, + ), + ).toBe(250); + }); + + it('returns the median gap so a single dropped sample does not skew it', () => { + const samples = [ + { ms: T0, value: 0 }, + { ms: T0 + 1000, value: 0 }, + { ms: T0 + 2000, value: 0 }, + { ms: T0 + 5000, value: 0 }, // dropped two rows + { ms: T0 + 6000, value: 0 }, + ]; + expect(estimateSampleIntervalMs(samples)).toBe(1000); + }); +}); + +describe('meanAcrossSeries', () => { + it('returns an empty array when there are no series or only empty series', () => { + expect(meanAcrossSeries([], 1000)).toEqual([]); + expect(meanAcrossSeries([[], []], 1000)).toEqual([]); + }); + + it('returns a single series as its own mean with count 1', () => { + const only = everySecond([10, 20]); + expect(meanAcrossSeries([only], 1000)).toEqual([ + { ms: only[0]!.ms, value: 10, count: 1 }, + { ms: only[1]!.ms, value: 20, count: 1 }, + ]); + }); + + it('aligns chips polled a few ms apart onto the reference timeline', () => { + // nvidia-smi writes chip rows ~12 ms apart within one 1 s poll. + const chip0 = everySecond([200, 400, 600]); + const chip1 = everySecond([300, 500, 700], 12); + const chip2 = everySecond([100, 300, 500], 25); + const out = meanAcrossSeries([chip0, chip1, chip2], 1000); + expect(out.map((p) => p.ms)).toEqual(chip0.map((p) => p.ms)); + expect(out.map((p) => p.value)).toEqual([200, 400, 600]); + expect(out.every((p) => p.count === 3)).toBe(true); + }); + + it('uses the longest series as the reference and skips chips with no sample in tolerance', () => { + const chip0 = everySecond([100, 100]); // shorter: stops early + const chip1 = everySecond([300, 300, 300, 300]); + const out = meanAcrossSeries([chip0, chip1], 500); + expect(out).toHaveLength(4); + expect(out[0]).toEqual({ ms: chip1[0]!.ms, value: 200, count: 2 }); + expect(out[1]).toEqual({ ms: chip1[1]!.ms, value: 200, count: 2 }); + // chip0 has nothing within 500 ms of t=2 s and t=3 s. + expect(out[2]).toEqual({ ms: chip1[2]!.ms, value: 300, count: 1 }); + expect(out[3]).toEqual({ ms: chip1[3]!.ms, value: 300, count: 1 }); + }); + + it('picks the nearest sample when a chip dropped a row', () => { + const chip0 = everySecond([0, 0, 0, 0]); + const chip1: TimedSample[] = [ + { ms: T0, value: 10 }, + // row at T0 + 1000 dropped + { ms: T0 + 2000, value: 30 }, + { ms: T0 + 3000, value: 40 }, + ]; + const out = meanAcrossSeries([chip0, chip1], 400); + expect(out.map((p) => p.count)).toEqual([2, 1, 2, 2]); + expect(out.map((p) => p.value)).toEqual([5, 0, 15, 20]); + }); + + it('treats a negative tolerance as exact-match only', () => { + const chip0 = everySecond([0, 0]); + const chip1 = everySecond([10, 10], 1); + expect(meanAcrossSeries([chip0, chip1], -1).every((p) => p.count === 1)).toBe(true); + expect(meanAcrossSeries([chip0, everySecond([10, 10])], -1).every((p) => p.count === 2)).toBe( + true, + ); + }); +}); diff --git a/packages/app/src/components/gpu-power/telemetry-smoothing.ts b/packages/app/src/components/gpu-power/telemetry-smoothing.ts index c4ee00e82..8c44c28ca 100644 --- a/packages/app/src/components/gpu-power/telemetry-smoothing.ts +++ b/packages/app/src/components/gpu-power/telemetry-smoothing.ts @@ -14,16 +14,25 @@ export interface TimedSample { export const SMOOTHING_WINDOWS_S = [10, 30, 60, 300] as const; export type SmoothingWindowS = (typeof SMOOTHING_WINDOWS_S)[number]; +export interface AggregatedSample extends TimedSample { + /** Number of chips that contributed a sample within tolerance. */ + count: number; +} + export type TelemetryDisplayMode = 'points' | 'rolling'; +/** Per-chip lines, one mean line across the visible chips, or both. */ +export type TelemetrySeriesMode = 'chips' | 'mean' | 'both'; export interface TelemetryDisplayState { mode: TelemetryDisplayMode; windowS: SmoothingWindowS; + series: TelemetrySeriesMode; } export const DEFAULT_TELEMETRY_DISPLAY: TelemetryDisplayState = { mode: 'points', windowS: 30, + series: 'chips', }; /** @@ -59,3 +68,66 @@ export function rollingTimeAverage( } return out; } + +/** + * Median gap between consecutive samples, in ms. Used as the alignment + * tolerance when averaging chips that were polled a few ms apart. Falls back + * to `fallbackMs` when fewer than two distinct timestamps exist. + */ +export function estimateSampleIntervalMs( + samples: readonly TimedSample[], + fallbackMs = 1000, +): number { + const gaps: number[] = []; + for (let i = 1; i < samples.length; i += 1) { + const gap = samples[i]!.ms - samples[i - 1]!.ms; + if (gap > 0) gaps.push(gap); + } + if (gaps.length === 0) return fallbackMs; + gaps.sort((a, b) => a - b); + return gaps[Math.floor(gaps.length / 2)]!; +} + +/** + * Mean across several chips at each timestamp of the reference chip (the one + * with the most samples, first on ties). For every reference sample each other + * chip contributes its nearest sample if that sample lies within + * `toleranceMs`; chips with no sample that close are left out of that mean + * rather than interpolated, and `count` records how many contributed. + * + * Every inner array must be sorted by `ms` ascending. + */ +export function meanAcrossSeries( + series: readonly (readonly TimedSample[])[], + toleranceMs: number, +): AggregatedSample[] { + const populated = series.filter((s) => s.length > 0); + if (populated.length === 0) return []; + let reference = populated[0]!; + for (const s of populated) if (s.length > reference.length) reference = s; + const others = populated.filter((s) => s !== reference); + const cursors: number[] = Array.from({ length: others.length }, () => 0); + const tolerance = Math.max(0, toleranceMs); + + return reference.map((ref) => { + let sum = ref.value; + let count = 1; + for (let k = 0; k < others.length; k += 1) { + const other = others[k]!; + let cursor = cursors[k]!; + while ( + cursor + 1 < other.length && + Math.abs(other[cursor + 1]!.ms - ref.ms) <= Math.abs(other[cursor]!.ms - ref.ms) + ) { + cursor += 1; + } + cursors[k] = cursor; + const candidate = other[cursor]!; + if (Math.abs(candidate.ms - ref.ms) <= tolerance) { + sum += candidate.value; + count += 1; + } + } + return { ms: ref.ms, value: sum / count, count }; + }); +} diff --git a/packages/app/src/lib/d3-chart/layers/lines.test.ts b/packages/app/src/lib/d3-chart/layers/lines.test.ts index 894e9d5f3..81b3a16dc 100644 --- a/packages/app/src/lib/d3-chart/layers/lines.test.ts +++ b/packages/app/src/lib/d3-chart/layers/lines.test.ts @@ -130,6 +130,29 @@ describe('renderLines', () => { } }); + it('uses getStrokeWidth per series and falls back to strokeWidth otherwise', () => { + const group = createMockGroup(); + const { xScale, yScale } = makeScales(); + renderLines( + group as any, + SAMPLE_LINES, + xScale, + yScale, + makeConfig({ + strokeWidth: 1.5, + getStrokeWidth: (key) => (key === 'seriesB' ? 3 : (undefined as unknown as number)), + }), + ); + + const widthByClass = Object.fromEntries( + group + .selectAll('.line-path') + .elements.map((el) => [el.attrs['class'], el.attrs['stroke-width']]), + ); + expect(widthByClass['line-path line-seriesA']).toBe(1.5); + expect(widthByClass['line-path line-seriesB']).toBe(3); + }); + it('generates valid d attribute from line generator', () => { const group = createMockGroup(); const { xScale, yScale } = makeScales(); diff --git a/packages/app/src/lib/d3-chart/layers/lines.ts b/packages/app/src/lib/d3-chart/layers/lines.ts index 075a1be71..885e0630d 100644 --- a/packages/app/src/lib/d3-chart/layers/lines.ts +++ b/packages/app/src/lib/d3-chart/layers/lines.ts @@ -9,6 +9,8 @@ export interface LineConfig { /** Optional per-series SVG dash pattern; return `none` for a solid line. */ getStrokeDasharray?: (key: string) => string; strokeWidth?: number; + /** Optional per-series stroke width; falls back to `strokeWidth`, then 2. */ + getStrokeWidth?: (key: string) => number; curve?: d3.CurveFactory; /** Return false to create gaps in the line (e.g., missing data points). */ isDefined?: (d: { x: number; y: number }) => boolean; @@ -61,7 +63,7 @@ export function renderLines( .attr('class', (d) => `line-path line-${d.key}`) .attr('stroke', (d) => config.getColor(d.key)) .attr('stroke-dasharray', (d) => config.getStrokeDasharray?.(d.key) ?? null) - .attr('stroke-width', config.strokeWidth ?? 2) + .attr('stroke-width', (d) => config.getStrokeWidth?.(d.key) ?? config.strokeWidth ?? 2) .attr('d', (d) => lineGenerator(d.points)); } From 1c234e3d7205d8a2521d1835fce5e793d8bbacc8 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 17 Sep 2026 14:04:40 -0500 Subject: [PATCH 004/103] feat(powerx): overlay decode throughput on the per-point telemetry chart Add an "Overlay decode throughput" switch (default off) to the per-point PowerX tab. When on, the point's decodeTps series from the trace server metrics is drawn over the telemetry in violet on its own right-hand axis, with a key-row entry and a row in every tooltip giving the nearest decode rate. The trace's startNs is an epoch-ns wall-clock timestamp, so the two series are aligned by absolute time; when a trace carries no wall-clock start the overlay falls back to the telemetry start and the note under the switch says so. Points without server metrics get an "unavailable" note instead of an empty overlay. The chart takes the overlay as an optional prop and renders it in a custom D3 layer that redraws on zoom and removes its own axis when toggled off. In rolling-average mode the overlay is smoothed with the same time window as the chip lines so both curves describe the same interval. Co-Authored-By: Claude Fable 5.1 --- .../components/gpu-power/GpuPowerChart.tsx | 180 ++++++++++++++++-- .../gpu-power/telemetry-smoothing.test.ts | 41 ++++ .../gpu-power/telemetry-smoothing.ts | 26 +++ .../agentic-point/power-telemetry-view.tsx | 79 +++++++- 4 files changed, 309 insertions(+), 17 deletions(-) diff --git a/packages/app/src/components/gpu-power/GpuPowerChart.tsx b/packages/app/src/components/gpu-power/GpuPowerChart.tsx index 17a9f4790..bcfe74e72 100644 --- a/packages/app/src/components/gpu-power/GpuPowerChart.tsx +++ b/packages/app/src/components/gpu-power/GpuPowerChart.tsx @@ -4,13 +4,17 @@ import * as d3 from 'd3'; import React, { useMemo } from 'react'; import { D3Chart } from '@/lib/d3-chart/D3Chart'; +import type { RenderContext } from '@/lib/d3-chart/D3Chart/types'; +import { CHART_TYPE, px } from '@/lib/d3-chart/typography'; import { useLocale } from '@/lib/use-locale'; import { DEFAULT_TELEMETRY_DISPLAY, estimateSampleIntervalMs, meanAcrossSeries, + nearestSample, rollingTimeAverage, type TelemetryDisplayState, + type TimedSample, } from './telemetry-smoothing'; import { type GpuMetricKey, @@ -35,6 +39,7 @@ const STRINGS = { rollingSuffix: (windowS: number) => `${windowS} s rolling avg`, meanChips: 'Mean of visible chips', meanCount: (n: number) => `${n} chips`, + rightAxis: 'right axis', }, zh: { empty: '暂无可显示的芯片指标数据。', @@ -48,9 +53,20 @@ const STRINGS = { rollingSuffix: (windowS: number) => `${windowS} 秒滚动平均`, meanChips: '可见芯片均值', meanCount: (n: number) => `${n} 个芯片`, + rightAxis: '右轴', }, } as const; +/** A second time series drawn over the telemetry on its own right-hand axis. */ +export interface TelemetryOverlaySeries { + key: string; + label: string; + unit: string; + color: string; + /** Absolute-time samples; the chart re-bases them onto its own t=0. */ + points: TimedSample[]; +} + interface ParsedPoint { seconds: number; /** Absolute sample time in ms; smoothing and alignment work in this space. */ @@ -74,6 +90,8 @@ interface GpuMetricsChartProps { maxPoints?: number; /** Raw samples vs. time-window rolling average. Defaults to raw samples. */ display?: TelemetryDisplayState; + /** Optional secondary series (e.g. decode throughput) on a right y-axis. */ + overlay?: TelemetryOverlaySeries | null; } function parseTimestamp(raw: string): Date | null { @@ -158,6 +176,80 @@ function colorFor(gpuIndex: number): string { } const CHART_ID = 'gpu-metrics-line'; const MARGIN = { top: 24, right: 20, bottom: 60, left: 60 }; +/** Room for the overlay's right axis ticks and rotated title. */ +const MARGIN_WITH_OVERLAY = { ...MARGIN, right: 84 }; + +interface OverlayPoint { + x: number; + y: number; +} + +function overlayYScale(points: OverlayPoint[], height: number): d3.ScaleLinear { + const max = d3.max(points, (p) => p.y) ?? 0; + return d3 + .scaleLinear() + .domain([0, max > 0 ? max * 1.05 : 1]) + .range([height, 0]) + .nice(); +} + +function overlayLine( + xScale: d3.ScaleLinear, + yScale: d3.ScaleLinear, +): d3.Line { + return d3 + .line() + .x((p) => xScale(p.x)) + .y((p) => yScale(p.y)) + .curve(d3.curveMonotoneX); +} + +/** + * Draw the overlay path inside the clipped zoom group and its axis in the + * unclipped root group. Both are removed first so toggling the overlay off + * (or re-rendering) never leaves a stale axis behind. + */ +function renderOverlay( + group: d3.Selection, + ctx: RenderContext, + overlay: TelemetryOverlaySeries | null | undefined, + points: OverlayPoint[], +): void { + group.selectAll('.telemetry-overlay').remove(); + ctx.layout.g.selectAll('.telemetry-overlay-axis').remove(); + if (!overlay || points.length === 0) return; + const xScale = ctx.xScale as d3.ScaleLinear; + const yScale = overlayYScale(points, ctx.height); + group + .append('path') + .attr('class', 'telemetry-overlay') + .attr('fill', 'none') + .attr('stroke', overlay.color) + .attr('stroke-width', 1.75) + .attr('opacity', 0.9) + .attr('pointer-events', 'none') + .attr('d', overlayLine(xScale, yScale)(points)); + + const axis = ctx.layout.g + .append('g') + .attr('class', 'telemetry-overlay-axis') + .attr('transform', `translate(${ctx.width},0)`) + .call(d3.axisRight(yScale).ticks(6).tickSize(4).tickFormat(d3.format('~s'))); + axis.select('.domain').attr('stroke', overlay.color); + axis.selectAll('.tick line').attr('stroke', overlay.color); + axis + .selectAll('.tick text') + .attr('fill', overlay.color) + .attr('font-size', px(CHART_TYPE.axisLabel)); + axis + .append('text') + .attr('class', 'telemetry-overlay-axis-label') + .attr('transform', `translate(${ctx.layout.margin.right - 14},${ctx.height / 2}) rotate(90)`) + .attr('text-anchor', 'middle') + .attr('fill', overlay.color) + .attr('font-size', px(CHART_TYPE.axisLabel)) + .text(`${overlay.label} (${overlay.unit})`); +} const GpuMetricsChart = React.memo( ({ @@ -169,6 +261,7 @@ const GpuMetricsChart = React.memo( caption, maxPoints, display = DEFAULT_TELEMETRY_DISPLAY, + overlay, }: GpuMetricsChartProps) => { const locale = useLocale(); const t = STRINGS[locale]; @@ -214,22 +307,51 @@ const GpuMetricsChart = React.memo( return result; }, [groupedData]); + // Overlay samples, smoothed with the same window as the chip lines when + // averaging so both series answer the same "how much over N seconds" + // question, then re-based onto the telemetry's t=0. + const overlaySamples = useMemo(() => { + if (!overlay) return []; + return rolling ? rollingTimeAverage(overlay.points, display.windowS * 1000) : overlay.points; + }, [overlay, rolling, display.windowS]); + const overlayPoints = useMemo( + () => overlaySamples.map((p) => ({ x: (p.ms - t0Ms) / 1000, y: p.value })), + [overlaySamples, t0Ms], + ); + const hasOverlay = Boolean(overlay) && overlayPoints.length > 0; + const hasMeanLine = groupedData.has(MEAN_INDEX); - const keyRow = hasMeanLine ? ( -
- - - {t.meanChips} - -
- ) : null; + const keyRow = + hasMeanLine || hasOverlay ? ( +
+ {hasMeanLine && ( + + + {t.meanChips} + + )} + {hasOverlay && overlay && ( + + + {overlay.label} ({overlay.unit} · {t.rightAxis}) + + )} +
+ ) : null; const resolvedCaption = caption || keyRow ? ( <> @@ -270,7 +392,7 @@ const GpuMetricsChart = React.memo( chartId={CHART_ID} data={allPoints} height={600} - margin={MARGIN} + margin={hasOverlay ? MARGIN_WITH_OVERLAY : MARGIN} watermark="logo" testId="gpu-metrics-chart-svg" grabCursor={true} @@ -339,6 +461,24 @@ const GpuMetricsChart = React.memo( maxPoints, }, }, + // Secondary series on a right-hand axis (e.g. decode throughput) + { + type: 'custom', + key: 'telemetry-overlay', + render: (group, ctx) => { + renderOverlay(group, ctx, hasOverlay ? overlay : null, overlayPoints); + }, + onZoom: (group, ctx) => { + if (!hasOverlay) return; + const xScale = ctx.newXScale as d3.ScaleLinear; + group + .select('.telemetry-overlay') + .attr( + 'd', + overlayLine(xScale, overlayYScale(overlayPoints, ctx.height))(overlayPoints), + ); + }, + }, ]} zoom={{ enabled: true, @@ -355,6 +495,13 @@ const GpuMetricsChart = React.memo( d.gpuIndex === MEAN_INDEX ? `${t.meanChips}${d.count ? ` · ${t.meanCount(d.count)}` : ''}` : `${t.chip} ${d.gpuIndex}`; + const overlayAt = hasOverlay ? nearestSample(overlaySamples, d.ms) : null; + const overlayRow = + overlay && overlayAt + ? `
${overlay.label}${sep} ${ + overlayAt.value >= 100 ? overlayAt.value.toFixed(0) : overlayAt.value.toFixed(1) + } ${overlay.unit}
` + : ''; return `
${isPinned ? `
${t.dismiss}
` : ''}
${title}
@@ -368,6 +515,7 @@ const GpuMetricsChart = React.memo(
${t.utilization}${sep} ${d.raw.gpuUtil}%
` : '' } + ${overlayRow}
`; }, getRulerX: (d, xScale) => (xScale as d3.ScaleLinear)(d.seconds), diff --git a/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts b/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts index a9d41298e..8865fc7fc 100644 --- a/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts +++ b/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts @@ -3,7 +3,9 @@ import { describe, expect, it } from 'vitest'; import { estimateSampleIntervalMs, meanAcrossSeries, + nearestSample, rollingTimeAverage, + toAbsoluteMs, type TimedSample, } from './telemetry-smoothing'; @@ -160,3 +162,42 @@ describe('meanAcrossSeries', () => { ); }); }); + +describe('toAbsoluteMs', () => { + it('offsets relative seconds from the given origin', () => { + // Trace startNs from a real point, as the API returns it. + const originMs = 1789508185611923200 / 1e6; + expect( + toAbsoluteMs( + [ + { t: 0, value: 0 }, + { t: 1.5, value: 42 }, + ], + originMs, + ), + ).toEqual([ + { ms: originMs, value: 0 }, + { ms: originMs + 1500, value: 42 }, + ]); + expect(toAbsoluteMs([], originMs)).toEqual([]); + }); +}); + +describe('nearestSample', () => { + const samples = everySecond([1, 2, 3, 4]); + + it('returns null for an empty series', () => { + expect(nearestSample([], T0)).toBeNull(); + }); + + it('returns the closest sample, clamping outside the series range', () => { + expect(nearestSample(samples, T0 - 5000)).toEqual(samples[0]); + expect(nearestSample(samples, T0 + 99_000)).toEqual(samples[3]); + expect(nearestSample(samples, T0 + 1400)).toEqual(samples[1]); + expect(nearestSample(samples, T0 + 1600)).toEqual(samples[2]); + }); + + it('prefers the earlier sample on an exact tie', () => { + expect(nearestSample(samples, T0 + 1500)).toEqual(samples[1]); + }); +}); diff --git a/packages/app/src/components/gpu-power/telemetry-smoothing.ts b/packages/app/src/components/gpu-power/telemetry-smoothing.ts index 8c44c28ca..91efcd64f 100644 --- a/packages/app/src/components/gpu-power/telemetry-smoothing.ts +++ b/packages/app/src/components/gpu-power/telemetry-smoothing.ts @@ -131,3 +131,29 @@ export function meanAcrossSeries( return { ms: ref.ms, value: sum / count, count }; }); } + +/** + * Convert a relative-seconds series (`t` from its own start) to absolute + * milliseconds so it can share an x-axis with wall-clock telemetry. + */ +export function toAbsoluteMs( + points: readonly { t: number; value: number }[], + originMs: number, +): TimedSample[] { + return points.map((p) => ({ ms: originMs + p.t * 1000, value: p.value })); +} + +/** Sample nearest to `ms` (earlier one on a tie), or null when the series is empty. */ +export function nearestSample(samples: readonly TimedSample[], ms: number): TimedSample | null { + if (samples.length === 0) return null; + let lo = 0; + let hi = samples.length - 1; + while (lo < hi) { + const mid = (lo + hi) >> 1; + if (samples[mid]!.ms < ms) lo = mid + 1; + else hi = mid; + } + const after = samples[lo]!; + const before = lo > 0 ? samples[lo - 1]! : after; + return Math.abs(before.ms - ms) <= Math.abs(after.ms - ms) ? before : after; +} diff --git a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx index f50c7d568..177b016d0 100644 --- a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx +++ b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx @@ -2,11 +2,15 @@ import { useMemo, useState } from 'react'; -import GpuMetricsChart, { GPU_COLORS } from '@/components/gpu-power/GpuPowerChart'; +import GpuMetricsChart, { + GPU_COLORS, + type TelemetryOverlaySeries, +} from '@/components/gpu-power/GpuPowerChart'; import GpuStatsTable from '@/components/gpu-power/GpuStatsTable'; import { TelemetryDisplayControls } from '@/components/gpu-power/TelemetryDisplayControls'; import { DEFAULT_TELEMETRY_DISPLAY, + toAbsoluteMs, type TelemetryDisplayState, } from '@/components/gpu-power/telemetry-smoothing'; import { @@ -20,6 +24,7 @@ import { Card } from '@/components/ui/card'; import ChartLegend from '@/components/ui/chart-legend'; import { Label } from '@/components/ui/label'; import { RetryableQueryError } from '@/components/ui/retryable-query-error'; +import { Switch } from '@/components/ui/switch'; import { Select, SelectContent, @@ -28,6 +33,7 @@ import { SelectValue, } from '@/components/ui/select'; import { useGpuMetricsPoint, type GpuMetricSeries } from '@/hooks/api/use-gpu-metrics-point'; +import { useTraceServerMetrics } from '@/hooks/api/use-trace-server-metrics'; import { track } from '@/lib/analytics'; import { useLocale } from '@/lib/use-locale'; @@ -50,6 +56,14 @@ const STRINGS = { chip: 'Chip', secondsUnit: 's', resetFilter: 'Show all chips', + overlayToggle: 'Overlay decode throughput', + decodeTps: 'Decode throughput', + overlayLoading: 'Loading server metrics…', + overlayError: 'Server metrics failed to load; the overlay is unavailable.', + overlayUnavailable: 'This point has no decode-throughput server metrics to overlay.', + overlayAligned: 'Decode throughput is aligned to the telemetry by wall-clock timestamps.', + overlayRelative: + 'The trace has no wall-clock timestamps, so decode throughput starts at the telemetry start (both at t=0).', }, zh: { loading: '正在加载 PowerX 遥测数据……', @@ -69,10 +83,19 @@ const STRINGS = { chip: '芯片', secondsUnit: '秒', resetFilter: '显示全部芯片', + overlayToggle: '叠加 decode 吞吐量', + decodeTps: 'Decode 吞吐量', + overlayLoading: '正在加载服务端指标……', + overlayError: '服务端指标加载失败,无法叠加显示。', + overlayUnavailable: '该数据点没有可叠加的 decode 吞吐量服务端指标。', + overlayAligned: 'Decode 吞吐量已按绝对时间戳与遥测数据对齐。', + overlayRelative: 'trace 缺少绝对时间戳,因此 decode 吞吐量与遥测数据均从各自的 t=0 开始对齐。', }, } as const; const VENDOR_LABEL: Record = { nvidia: 'nvidia-smi', amd: 'amd-smi' }; +/** Violet: outside the Tableau10 chip palette and the foreground mean line. */ +const OVERLAY_COLOR = '#8b5cf6'; interface Props { id: number; @@ -143,6 +166,35 @@ export function PowerTelemetryView({ id, enabled }: Props) { const [isLegendExpanded, setIsLegendExpanded] = useState(true); const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); + // Decode-throughput overlay: fetched only once the switch is on. + const [overlayEnabled, setOverlayEnabled] = useState(false); + const metricsQuery = useTraceServerMetrics(id, enabled && overlayEnabled); + const serverMetrics = metricsQuery.data; + const decodeTps = serverMetrics?.decodeTps ?? []; + // Trace timeslices carry epoch-ns starts, so both series can share wall-clock + // time. A zero startNs means the trace only has relative time. + const overlayAbsolute = Boolean(serverMetrics && serverMetrics.startNs > 0); + const overlay = useMemo(() => { + if (!overlayEnabled || !serverMetrics || !selectedSeries || decodeTps.length === 0) return null; + const originMs = overlayAbsolute + ? serverMetrics.startNs / 1e6 + : new Date(selectedSeries.startedAt).getTime(); + return { + key: 'decodeTps', + label: t.decodeTps, + unit: 'tok/s', + color: OVERLAY_COLOR, + points: toAbsoluteMs(decodeTps, originMs), + }; + }, [overlayEnabled, serverMetrics, selectedSeries, decodeTps, overlayAbsolute, t.decodeTps]); + const overlayNote = ((): string | null => { + if (!overlayEnabled) return null; + if (metricsQuery.isLoading) return t.overlayLoading; + if (metricsQuery.isError) return t.overlayError; + if (!overlay) return t.overlayUnavailable; + return overlayAbsolute ? t.overlayAligned : t.overlayRelative; + })(); + if (!enabled) return null; if (query.isLoading) { @@ -276,6 +328,30 @@ export function PowerTelemetryView({ id, enabled }: Props) { idPrefix="power-telemetry-display" className="mt-3 border-t border-border/60 pt-3" /> +
+
+ { + track('inference_agentic_power_overlay_toggled', { id, enabled: checked }); + setOverlayEnabled(checked); + }} + /> + +
+ {overlayNote && ( + + {overlayNote} + + )} +
@@ -286,6 +362,7 @@ export function PowerTelemetryView({ id, enabled }: Props) { artifactName={selectedSeries.artifactName} maxPoints={2000} display={display} + overlay={overlay} legendElement={ Date: Thu, 17 Sep 2026 14:12:59 -0500 Subject: [PATCH 005/103] feat(powerx): default telemetry charts to rolling average At 1 s cadence the raw per-chip points read as noise, so both PowerX surfaces now open in rolling-average mode (30 s window); Points remains one click away in the Display control. Co-Authored-By: Claude Fable 5.1 --- packages/app/src/components/gpu-power/telemetry-smoothing.ts | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/packages/app/src/components/gpu-power/telemetry-smoothing.ts b/packages/app/src/components/gpu-power/telemetry-smoothing.ts index 91efcd64f..a9ec9a307 100644 --- a/packages/app/src/components/gpu-power/telemetry-smoothing.ts +++ b/packages/app/src/components/gpu-power/telemetry-smoothing.ts @@ -29,8 +29,9 @@ export interface TelemetryDisplayState { series: TelemetrySeriesMode; } +/** Rolling average is the default: at 1 s cadence the raw points read as noise. */ export const DEFAULT_TELEMETRY_DISPLAY: TelemetryDisplayState = { - mode: 'points', + mode: 'rolling', windowS: 30, series: 'chips', }; From ab479c8cae0f3c0294b9194596e3a5804b7d9d23 Mon Sep 17 00:00:00 2001 From: Cam Quilici Date: Thu, 17 Sep 2026 14:24:43 -0500 Subject: [PATCH 006/103] feat(powerx): choose any available server metric as the telemetry overlay Replace the decode-only overlay switch on the per-point PowerX tab with a single-select menu of server metrics. The menu lists only the series the point's trace actually reports (decode/prefill throughput, KV and host KV cache utilization, prefix cache hit rate and hits, queue depth), defaults to "None", and is disabled when the trace has no server metrics. Co-Authored-By: Claude Fable 5.1 --- .../agentic-point/overlay-sources.test.ts | 61 ++++++++++++ .../agentic-point/overlay-sources.ts | 83 ++++++++++++++++ .../agentic-point/power-telemetry-view.tsx | 98 +++++++++++-------- 3 files changed, 202 insertions(+), 40 deletions(-) create mode 100644 packages/app/src/components/inference/agentic-point/overlay-sources.test.ts create mode 100644 packages/app/src/components/inference/agentic-point/overlay-sources.ts diff --git a/packages/app/src/components/inference/agentic-point/overlay-sources.test.ts b/packages/app/src/components/inference/agentic-point/overlay-sources.test.ts new file mode 100644 index 000000000..04bc1d44a --- /dev/null +++ b/packages/app/src/components/inference/agentic-point/overlay-sources.test.ts @@ -0,0 +1,61 @@ +import { describe, expect, it } from 'vitest'; + +import type { TraceServerMetrics } from '@/hooks/api/use-trace-server-metrics'; + +import { availableOverlaySources, OVERLAY_SOURCES } from './overlay-sources'; + +const base: TraceServerMetrics = { + meta: {} as TraceServerMetrics['meta'], + startNs: 1_789_508_185_611_923_200, + endNs: 1_789_512_197_546_315_776, + durationS: 4011.9, + timeslicesCount: 3, + kvCacheUsage: [], + prefixCacheHitRate: [], + queueDepth: [], + promptTokensBySource: {}, + prefillTps: [], + decodeTps: [], + prefixCacheHitsTps: [], + hostKvCacheUsage: [], + kvCacheUsageByEngine: [], + kvCachePoolTokens: null, + metricSources: [], +}; + +describe('availableOverlaySources', () => { + it('returns nothing without metrics or when every series is empty', () => { + expect(availableOverlaySources(null)).toEqual([]); + expect(availableOverlaySources(base)).toEqual([]); + }); + + it('offers only non-empty series, in menu order', () => { + const metrics: TraceServerMetrics = { + ...base, + queueDepth: [{ t: 0, running: 3, waiting: 2, total: 5 }], + decodeTps: [{ t: 0, value: 1200 }], + }; + expect(availableOverlaySources(metrics).map((s) => s.key)).toEqual(['decodeTps', 'queueDepth']); + }); + + it('converts ratios to percent and queue depth to the total', () => { + const metrics: TraceServerMetrics = { + ...base, + kvCacheUsage: [{ t: 1, value: 0.42 }], + hostKvCacheUsage: [{ t: 1, value: 0.05 }], + prefixCacheHitRate: [{ t: 1, value: 1 }], + queueDepth: [{ t: 1, running: 3, waiting: 2, total: 5 }], + }; + const byKey = Object.fromEntries(OVERLAY_SOURCES.map((s) => [s.key, s.points(metrics)])); + expect(byKey.kvCacheUsage).toEqual([{ t: 1, value: 42 }]); + expect(byKey.hostKvCacheUsage[0]!.value).toBeCloseTo(5, 9); + expect(byKey.prefixCacheHitRate).toEqual([{ t: 1, value: 100 }]); + expect(byKey.queueDepth).toEqual([{ t: 1, value: 5 }]); + expect(byKey.decodeTps).toEqual([]); + }); + + it('gives every source a distinct key and color', () => { + expect(new Set(OVERLAY_SOURCES.map((s) => s.key)).size).toBe(OVERLAY_SOURCES.length); + expect(new Set(OVERLAY_SOURCES.map((s) => s.color)).size).toBe(OVERLAY_SOURCES.length); + }); +}); diff --git a/packages/app/src/components/inference/agentic-point/overlay-sources.ts b/packages/app/src/components/inference/agentic-point/overlay-sources.ts new file mode 100644 index 000000000..029274f25 --- /dev/null +++ b/packages/app/src/components/inference/agentic-point/overlay-sources.ts @@ -0,0 +1,83 @@ +import type { TraceServerMetrics } from '@/hooks/api/use-trace-server-metrics'; +import type { Locale } from '@/lib/i18n'; + +/** One server-metric series that can be drawn over the chip telemetry chart. */ +export interface OverlaySource { + key: string; + label: { en: string; zh: string }; + unit: string; + color: string; + /** Seconds-from-trace-start samples, already in display units. */ + points: (metrics: TraceServerMetrics) => { t: number; value: number }[]; +} + +const percent = (points: readonly { t: number; value: number }[]) => + points.map((p) => ({ t: p.t, value: p.value * 100 })); + +/** + * Menu of overlay candidates, in display order. Only sources whose series is + * non-empty for the point are offered (see `availableOverlaySources`). + */ +export const OVERLAY_SOURCES: readonly OverlaySource[] = [ + { + key: 'decodeTps', + label: { en: 'Decode throughput', zh: 'Decode 吞吐量' }, + unit: 'tok/s', + color: '#8b5cf6', + points: (m) => m.decodeTps, + }, + { + key: 'prefillTps', + label: { en: 'Prefill throughput', zh: 'Prefill 吞吐量' }, + unit: 'tok/s', + color: '#06b6d4', + points: (m) => m.prefillTps, + }, + { + key: 'kvCacheUsage', + label: { en: 'KV cache utilization', zh: 'KV cache 利用率' }, + unit: '%', + color: '#f59e0b', + points: (m) => percent(m.kvCacheUsage), + }, + { + key: 'hostKvCacheUsage', + label: { en: 'Host KV cache utilization', zh: '主机 KV cache 利用率' }, + unit: '%', + color: '#d97706', + points: (m) => percent(m.hostKvCacheUsage), + }, + { + key: 'prefixCacheHitRate', + label: { en: 'Prefix cache hit rate', zh: 'Prefix cache 命中率' }, + unit: '%', + color: '#10b981', + points: (m) => percent(m.prefixCacheHitRate), + }, + { + key: 'prefixCacheHitsTps', + label: { en: 'Prefix cache hits', zh: 'Prefix cache 命中量' }, + unit: 'tok/s', + color: '#14b8a6', + points: (m) => m.prefixCacheHitsTps, + }, + { + key: 'queueDepth', + label: { en: 'Queue depth (running + waiting)', zh: '队列深度(运行中 + 等待中)' }, + unit: 'req', + color: '#ec4899', + points: (m) => m.queueDepth.map((p) => ({ t: p.t, value: p.total })), + }, +]; + +/** Sources that have at least one sample for this point, in menu order. */ +export function availableOverlaySources( + metrics: TraceServerMetrics | null | undefined, +): OverlaySource[] { + if (!metrics) return []; + return OVERLAY_SOURCES.filter((source) => source.points(metrics).length > 0); +} + +export function overlaySourceLabel(source: OverlaySource, locale: Locale): string { + return source.label[locale]; +} diff --git a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx index 177b016d0..967ca41e9 100644 --- a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx +++ b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx @@ -24,7 +24,6 @@ import { Card } from '@/components/ui/card'; import ChartLegend from '@/components/ui/chart-legend'; import { Label } from '@/components/ui/label'; import { RetryableQueryError } from '@/components/ui/retryable-query-error'; -import { Switch } from '@/components/ui/switch'; import { Select, SelectContent, @@ -34,6 +33,8 @@ import { } from '@/components/ui/select'; import { useGpuMetricsPoint, type GpuMetricSeries } from '@/hooks/api/use-gpu-metrics-point'; import { useTraceServerMetrics } from '@/hooks/api/use-trace-server-metrics'; + +import { availableOverlaySources, overlaySourceLabel } from './overlay-sources'; import { track } from '@/lib/analytics'; import { useLocale } from '@/lib/use-locale'; @@ -56,14 +57,14 @@ const STRINGS = { chip: 'Chip', secondsUnit: 's', resetFilter: 'Show all chips', - overlayToggle: 'Overlay decode throughput', - decodeTps: 'Decode throughput', + overlayToggle: 'Overlay server metric', + overlayNone: 'None', overlayLoading: 'Loading server metrics…', - overlayError: 'Server metrics failed to load; the overlay is unavailable.', - overlayUnavailable: 'This point has no decode-throughput server metrics to overlay.', - overlayAligned: 'Decode throughput is aligned to the telemetry by wall-clock timestamps.', + overlayError: 'Server metrics failed to load; overlays are unavailable.', + overlayUnavailable: 'This point has no server-metric series to overlay.', + overlayAligned: 'The overlay is aligned to the telemetry by wall-clock timestamps.', overlayRelative: - 'The trace has no wall-clock timestamps, so decode throughput starts at the telemetry start (both at t=0).', + 'The trace has no wall-clock timestamps, so the overlay and the telemetry are both aligned at their own t=0.', }, zh: { loading: '正在加载 PowerX 遥测数据……', @@ -83,19 +84,18 @@ const STRINGS = { chip: '芯片', secondsUnit: '秒', resetFilter: '显示全部芯片', - overlayToggle: '叠加 decode 吞吐量', - decodeTps: 'Decode 吞吐量', + overlayToggle: '叠加服务端指标', + overlayNone: '无', overlayLoading: '正在加载服务端指标……', overlayError: '服务端指标加载失败,无法叠加显示。', - overlayUnavailable: '该数据点没有可叠加的 decode 吞吐量服务端指标。', - overlayAligned: 'Decode 吞吐量已按绝对时间戳与遥测数据对齐。', - overlayRelative: 'trace 缺少绝对时间戳,因此 decode 吞吐量与遥测数据均从各自的 t=0 开始对齐。', + overlayUnavailable: '该数据点没有可叠加的服务端指标序列。', + overlayAligned: '叠加曲线已按绝对时间戳与遥测数据对齐。', + overlayRelative: 'trace 缺少绝对时间戳,因此叠加曲线与遥测数据均从各自的 t=0 开始对齐。', }, } as const; const VENDOR_LABEL: Record = { nvidia: 'nvidia-smi', amd: 'amd-smi' }; /** Violet: outside the Tableau10 chip palette and the foreground mean line. */ -const OVERLAY_COLOR = '#8b5cf6'; interface Props { id: number; @@ -166,32 +166,37 @@ export function PowerTelemetryView({ id, enabled }: Props) { const [isLegendExpanded, setIsLegendExpanded] = useState(true); const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); - // Decode-throughput overlay: fetched only once the switch is on. - const [overlayEnabled, setOverlayEnabled] = useState(false); - const metricsQuery = useTraceServerMetrics(id, enabled && overlayEnabled); + // Server-metric overlay. The series are fetched as soon as the tab opens so + // the menu can list exactly the metrics this point has; one source at a time. + const metricsQuery = useTraceServerMetrics(id, enabled); const serverMetrics = metricsQuery.data; - const decodeTps = serverMetrics?.decodeTps ?? []; + const overlaySources = useMemo(() => availableOverlaySources(serverMetrics), [serverMetrics]); + const [overlaySelection, setOverlaySelection] = useState<{ id: number; key: string } | null>( + null, + ); + const overlayKey = overlaySelection?.id === id ? overlaySelection.key : 'none'; + const overlaySource = overlaySources.find((source) => source.key === overlayKey) ?? null; // Trace timeslices carry epoch-ns starts, so both series can share wall-clock // time. A zero startNs means the trace only has relative time. const overlayAbsolute = Boolean(serverMetrics && serverMetrics.startNs > 0); const overlay = useMemo(() => { - if (!overlayEnabled || !serverMetrics || !selectedSeries || decodeTps.length === 0) return null; + if (!overlaySource || !serverMetrics || !selectedSeries) return null; const originMs = overlayAbsolute ? serverMetrics.startNs / 1e6 : new Date(selectedSeries.startedAt).getTime(); return { - key: 'decodeTps', - label: t.decodeTps, - unit: 'tok/s', - color: OVERLAY_COLOR, - points: toAbsoluteMs(decodeTps, originMs), + key: overlaySource.key, + label: overlaySourceLabel(overlaySource, locale), + unit: overlaySource.unit, + color: overlaySource.color, + points: toAbsoluteMs(overlaySource.points(serverMetrics), originMs), }; - }, [overlayEnabled, serverMetrics, selectedSeries, decodeTps, overlayAbsolute, t.decodeTps]); + }, [overlaySource, serverMetrics, selectedSeries, overlayAbsolute, locale]); const overlayNote = ((): string | null => { - if (!overlayEnabled) return null; if (metricsQuery.isLoading) return t.overlayLoading; if (metricsQuery.isError) return t.overlayError; - if (!overlay) return t.overlayUnavailable; + if (overlaySources.length === 0) return t.overlayUnavailable; + if (!overlay) return null; return overlayAbsolute ? t.overlayAligned : t.overlayRelative; })(); @@ -328,24 +333,37 @@ export function PowerTelemetryView({ id, enabled }: Props) { idPrefix="power-telemetry-display" className="mt-3 border-t border-border/60 pt-3" /> -
-
- { - track('inference_agentic_power_overlay_toggled', { id, enabled: checked }); - setOverlayEnabled(checked); +
+
+ +
{overlayNote && ( {overlayNote} From 132f0b5a83334ad10543e23400e00cf527d0dd4a Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 17 Sep 2026 14:20:27 -0700 Subject: [PATCH 007/103] fix: register gpu-metrics-point route and refresh gpu-metrics catalog digest MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Register GET /api/v1/gpu-metrics-point as a page-owned BFF in the API route catalog and refresh the /api/gpu-metrics digest after the DB-first digest read landed in d5e90c40. Neither route is part of the public API reference, so the OpenAPI document and human reference are unchanged; the exclusion reason for /api/gpu-metrics now describes the stored-digest-then-live-artifact behavior. 中文:在 API 路由目录中登记 GET /api/v1/gpu-metrics-point(页面专用 BFF),并在 d5e90c40 引入 DB 优先读取后刷新 /api/gpu-metrics 的摘要。两条路由都不属于公开 API 参考,OpenAPI 文档与人工参考无需改动;/api/gpu-metrics 的排除说明改为描述 "已入库摘要优先、否则回退实时制品"的行为。 --- packages/app/src/lib/api-route-catalog.ts | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/packages/app/src/lib/api-route-catalog.ts b/packages/app/src/lib/api-route-catalog.ts index ad54bf6a8..6f090ad9f 100644 --- a/packages/app/src/lib/api-route-catalog.ts +++ b/packages/app/src/lib/api-route-catalog.ts @@ -68,10 +68,10 @@ export const apiRouteCatalog = [ method: 'GET', classification: 'ui-artifact-read', exclusionReason: { - en: 'UI-only live GPU metric artifact lookup; its run artifact shape is not a stable public contract.', - zh: '仅供界面读取实时 GPU 指标制品;其运行制品结构不是稳定的公开契约。', + en: 'UI-only PowerX explorer read for one run: the ingest-time telemetry digest when stored, otherwise the live GPU metric artifacts. Its payload shape is not a stable public contract.', + zh: '仅供 PowerX 探索界面按 run 读取:已入库时返回 ingest 阶段生成的 telemetry 摘要,否则回退到实时 GPU 指标制品。其返回结构不是稳定的公开契约。', }, - sourceSha256: '28e6cee4d67396ee8ea2e5a7e18271c6ee86228c33f33a20bf573f3a601ba8ed', + sourceSha256: '01d604d77ea73e252f0934f88d14d0e229a803d1b9b1d72bd9756658c506b349', }, { source: 'src/app/api/openapi.json/route.ts', @@ -327,6 +327,17 @@ export const apiRouteCatalog = [ operationId: 'list-reliability', sourceSha256: 'ce1c5db78b47548beb77a69797f10fb33853cde01cea5c44675c8ad3519bcf20', }, + { + source: 'src/app/api/v1/gpu-metrics-point/route.ts', + path: '/api/v1/gpu-metrics-point', + method: 'GET', + classification: 'page-bff', + exclusionReason: { + en: 'Agentic point-detail BFF returning the PowerX telemetry series and per-GPU digest linked to one benchmark point; coupled to the PowerX tab implementation.', + zh: '智能体数据点详情页专用 BFF;返回与单个基准测试数据点关联的 PowerX telemetry 序列及每 GPU 统计摘要,与 PowerX 标签页实现紧密耦合。', + }, + sourceSha256: '3929581f54058183344d59a8f6127db3ef1c8487cad26c73faa93b00fe75a82d', + }, { source: 'src/app/api/v1/request-chart-data/route.ts', path: '/api/v1/request-chart-data', From 70d4cab758889f19c7b5a10e87373d0b9428f5d9 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 17 Sep 2026 14:27:05 -0700 Subject: [PATCH 008/103] docs: correct why gpu_metrics telemetry backfill is bounded to 90 days MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The pipeline doc blamed an artifact-name filter in the GCS backup. The backup keeps every artifact name; it only mirrors scheduled and main-branch runs, so PR sweeps and manual dispatches - nearly every telemetry-bearing run - are never copied, and our own GCS reader ignores non-bmk_/server_logs_ objects anyway. Also note that the multinode template uploads no gpu_metrics_ artifact, so multinode and disaggregated points never get a telemetry series. 中文:修正数据管线文档中 gpu_metrics telemetry 回填受 90 天限制的原因说明。 GCS 备份并不按 artifact 名过滤,而是只镜像定时任务和 main 分支的运行,PR sweep 与手动触发的运行从不被复制;应用自身的 GCS reader 也只读取 bmk_/server_logs_ 对象。同时注明多节点模板不上传 gpu_metrics_ artifact,多节点与 disagg 数据点 不会有 telemetry 序列。 --- docs/data-pipeline.md | 24 ++++++++++++++++-------- 1 file changed, 16 insertions(+), 8 deletions(-) diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index ebafd3005..eaee38601 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -525,12 +525,15 @@ Reads are **permanently tolerant**: `queries/benchmarks.ts` selects the columns ### PowerX Telemetry Digest (`gpu_metric_*`, migration 016) -Every benchmark job samples `nvidia-smi` / `amd-smi` once per second for its -whole lifetime and uploads the CSV as `gpu_metrics_` next to -`bmk_` (agentic jobs: `bmk_agentic_`, still paired by the bare -suffix). The PowerX explorer used to download and parse those artifacts from -GitHub on every request and lost them after GitHub's 90-day retention. CI ingest -now digests them at ingest time, in the same step that links server logs: +Every single-node benchmark job (`benchmark-tmpl.yml`) samples `nvidia-smi` / +`amd-smi` once per second for its whole lifetime and uploads the CSV as +`gpu_metrics_` next to `bmk_` (agentic jobs: `bmk_agentic_`, +still paired by the bare suffix). The multinode template uploads no `gpu_metrics_` +artifact — its telemetry travels only inside `power_audit_` — so multinode and +disaggregated points have no series here and their per-point PowerX tab stays +empty. The PowerX explorer used to download and parse the artifacts from GitHub +on every request and lost them after GitHub's 90-day retention. CI ingest now +digests them at ingest time, in the same step that links server logs: - `gpu_metric_series` — one row per (workflow run, artifact, CSV path): vendor, CSV sha256, sample count, GPU count, recorded window, median cadence, and the @@ -555,8 +558,13 @@ interval is a reader concern; the raw series deliberately includes server start-up and warm-up so both phases can be inspected. `bun run admin:db:backfill-gpu-metrics --all --yes` attaches telemetry for runs -ingested before this migration. The GCS backup only mirrors `bmk_`/`server_logs_` -uploads, so the reachable history is bounded by GitHub's 90-day retention. +ingested before this migration. The reachable history is bounded by GitHub's +90-day artifact retention (the upload step sets no `retention-days`) because the +GCS backup, which keeps every artifact name, only mirrors `schedule` and `push` +runs on `main`; PR sweeps and manual dispatches — nearly every telemetry-bearing +run — are never copied. Our own GCS reader (`lib/gcs-artifacts.ts`) additionally +ignores everything but `bmk_`/`server_logs_` objects, so widening the mirror's run +filter would also need a reader change before backfill could use it. Readers: `/api/gpu-metrics?runId=` serves the digest when the run is stored and falls back to live GitHub artifacts otherwise (in-progress runs), and From 87b5ebe9271cd796008c7aa277556b1e99fec3ef Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Thu, 17 Sep 2026 14:30:17 -0700 Subject: [PATCH 009/103] test: cover gpu-metrics backfill, queries, and point lookup MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 30 PGlite-backed tests over the PowerX telemetry ingest code that had none. - queries/gpu-metrics: null on unknown run and on a run with zero series, full run payload shape, includeSamples=false, latest-attempt selection, multinode point fan-out, availability map, and a characterization of the `?? 0` sample defaulting (a dropped reading reports 0 W). - lib/benchmark-result-lookup: natural-key match scoped to run attempt and config, offload_mode exact match vs. unique fallback vs. ambiguous drop, agentic null isl/osl, and artifact JSON mapping. - backfill-gpu-metrics: candidate selection driven through the real CLI main against a migrated PGlite database, with GitHub listing stubbed. Pins the 90-day default --since, the resume hazard (runs with stored series are skipped without --force), --run bypassing both filters, --from-run, --limit, and that --shard-count/--shard-index do not partition the set. No production source was changed. 中文:为此前完全没有测试的 PowerX 遥测 ingest 代码补充 30 个基于 PGlite 的单元 测试,覆盖 gpu-metrics 读取查询、backfill 候选筛选与 benchmark 点位匹配。要点: 读取查询在 run 不存在或没有 series 时返回 null,多节点点位返回全部关联 series, 并固化了 `?? 0` 默认值行为(采集中断的样本会显示为 0 W,与真实的 0 W 无法区分); 点位匹配按 run attempt 与完整 config 自然键限定,并区分 offload_mode 精确匹配、 唯一回退与歧义丢弃;backfill 测试通过真实 main 驱动候选筛选,固化默认 90 天 --since 窗口、--force 之前会跳过已有 series 的 run(断点续跑隐患)、--run 同时 绕过日期与 series 过滤,以及 --shard-count/--shard-index 实际不分片的现状。 未修改任何生产代码。 --- packages/db/src/backfill-gpu-metrics.test.ts | 192 ++++++++++++ .../src/lib/benchmark-result-lookup.test.ts | 219 +++++++++++++ packages/db/src/queries/gpu-metrics.test.ts | 295 ++++++++++++++++++ 3 files changed, 706 insertions(+) create mode 100644 packages/db/src/backfill-gpu-metrics.test.ts create mode 100644 packages/db/src/lib/benchmark-result-lookup.test.ts create mode 100644 packages/db/src/queries/gpu-metrics.test.ts diff --git a/packages/db/src/backfill-gpu-metrics.test.ts b/packages/db/src/backfill-gpu-metrics.test.ts new file mode 100644 index 000000000..e718fdd4d --- /dev/null +++ b/packages/db/src/backfill-gpu-metrics.test.ts @@ -0,0 +1,192 @@ +/** + * Candidate-selection tests for the gpu_metrics backfill CLI. + * + * The script owns its `sql` handle and calls `runBackfillMain` at import time, + * so the seam here is the module boundary: `createAdminSql` is redirected at a + * PGlite database that has the real migrations applied, `runBackfillMain` + * captures `main` instead of running it, and `listRunArtifacts` records which + * runs the CLI actually reached. Every candidate query runs for real; nothing + * touches GitHub. + */ + +import fs from 'node:fs'; + +import { PGlite } from '@electric-sql/pglite'; +import type postgres from 'postgres'; +import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest'; + +import type * as DbUtilsModule from './etl/db-utils'; +import type * as BackfillRunnerModule from './lib/backfill-runner'; +import type * as GithubArtifactsModule from './lib/github-artifacts'; +import type { ArtifactMeta } from './lib/github-artifacts'; + +type Sql = postgres.Sql; +let db: PGlite; +let sql: Sql; +let main: () => Promise; +const listedRunIds: string[] = []; +const originalArgv = process.argv; + +vi.mock('./etl/db-utils.js', async (importOriginal) => ({ + ...(await importOriginal()), + createAdminSql: () => sql, +})); + +vi.mock('./lib/github-artifacts.js', async (importOriginal) => ({ + ...(await importOriginal()), + listRunArtifacts: (_repository: string, runId: string): Promise => { + listedRunIds.push(runId); + return Promise.resolve([]); + }, +})); + +vi.mock('./lib/backfill-runner.js', async (importOriginal) => ({ + ...(await importOriginal()), + runBackfillMain: (_name: string, _sql: Sql, entry: () => Promise) => { + main = entry; + }, +})); + +function queryClient(database: Pick) { + const client = async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await database.query(query, values); + return result.rows; + }; + return Object.assign(client, { json: JSON.stringify, array: (value: unknown) => value }); +} + +const daysAgo = (days: number): string => + new Date(Date.now() - days * 24 * 60 * 60 * 1000).toISOString().slice(0, 10); + +const FRESH_NO_SERIES = 34000000001; +const FRESH_WITH_SERIES = 34000000002; +const STALE_WITH_SERIES = 34000000003; +const FRESH_NO_BENCHMARKS = 34000000004; + +/** Run the CLI in --dry-run and report which GitHub run ids it selected, in order. */ +async function selectedRuns(...args: string[]): Promise { + listedRunIds.length = 0; + process.argv = ['bun', 'src/backfill-gpu-metrics.ts', '--dry-run', ...args]; + await main(); + return listedRunIds.map(Number); +} + +beforeAll(async () => { + db = await PGlite.create(); + for (const name of ['001_initial_schema.sql', '016_gpu_metrics.sql']) { + await db.exec(fs.readFileSync(new URL(`../migrations/${name}`, import.meta.url), 'utf8')); + } + sql = queryClient(db) as unknown as Sql; + await import('./backfill-gpu-metrics'); + expect(typeof main).toBe('function'); +}, 20_000); + +afterAll(async () => { + process.argv = originalArgv; + await db?.close(); +}); + +/** + * Four runs covering each axis of the candidate filter: fresh/stale relative to + * GitHub's 90-day artifact retention, with/without an already-stored series, and + * one run that produced no benchmark rows at all. + */ +beforeEach(async () => { + listedRunIds.length = 0; + vi.spyOn(console, 'log').mockImplementation(() => {}); + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, html_url, created_at, date) + VALUES (1, ${FRESH_NO_SERIES}, 1, 'Run Sweep', null, now(), '${daysAgo(10)}'), + (2, ${FRESH_WITH_SERIES}, 1, 'Run Sweep', null, now(), '${daysAgo(10)}'), + (3, ${STALE_WITH_SERIES}, 1, 'Run Sweep', null, now(), '${daysAgo(100)}'), + (4, ${FRESH_NO_BENCHMARKS}, 1, 'Run Sweep', null, now(), '${daysAgo(10)}'); + + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, 8, 8, 8, 8); + + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, + isl, osl, conc, metrics) + VALUES (10, 1, 1, 'single_turn', '${daysAgo(10)}', 8192, 1024, 32, '{}'), + (11, 2, 1, 'single_turn', '${daysAgo(10)}', 8192, 1024, 32, '{}'), + (12, 3, 1, 'single_turn', '${daysAgo(100)}', 8192, 1024, 32, '{}'); + + INSERT INTO gpu_metric_series (workflow_run_id, artifact_name, config_key, file_name, + vendor, csv_sha256, sample_count, gpu_count, started_at, ended_at) + VALUES (2, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', 'gpu_metrics.csv', 'nvidia', 'sha-2', + 1, 8, now(), now()), + (3, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', 'gpu_metrics.csv', 'nvidia', 'sha-3', + 1, 8, now(), now());`); +}); + +describe('candidate selection', () => { + it('defaults --since to GitHub 90-day artifact retention window', async () => { + // --force removes the series filter, isolating the date cutoff. The 100-day-old + // run is unreachable on GitHub, so it must not be queued. + expect(await selectedRuns('--all', '--force')).toEqual([FRESH_WITH_SERIES, FRESH_NO_SERIES]); + }); + + it('reaches past the retention window when --since is given explicitly', async () => { + expect(await selectedRuns('--all', '--force', '--since', '2020-01-01')).toEqual([ + FRESH_WITH_SERIES, + FRESH_NO_SERIES, + STALE_WITH_SERIES, + ]); + }); + + it('resume hazard: a run that stored any series is skipped on a rerun without --force', async () => { + // An interrupted backfill leaves a run with one of its several artifacts + // stored. Re-invoking --all silently treats that run as finished. + expect(await selectedRuns('--all')).toEqual([FRESH_NO_SERIES]); + }); + + it('--run bypasses both the retention cutoff and the already-has-series filter', async () => { + expect(await selectedRuns('--run', String(STALE_WITH_SERIES))).toEqual([STALE_WITH_SERIES]); + }); + + it('--run still requires the run to have benchmark rows to attach telemetry to', async () => { + expect(await selectedRuns('--run', String(FRESH_NO_BENCHMARKS))).toEqual([]); + }); + + it('--from-run raises the GitHub run id floor', async () => { + expect(await selectedRuns('--all', '--force', '--from-run', String(FRESH_WITH_SERIES))).toEqual( + [FRESH_WITH_SERIES], + ); + }); + + it('--limit truncates the ordered candidate list', async () => { + expect(await selectedRuns('--all', '--force', '--limit', '1')).toEqual([FRESH_WITH_SERIES]); + }); + + it('accepts --shard-count/--shard-index but does not partition the candidate set', async () => { + // Unlike the other sharded backfills, this script never applies the shard to + // its query, so every shard process would download the same artifacts. + const unsharded = await selectedRuns('--all', '--force'); + for (const shardIndex of ['0', '1', '2', '3']) { + expect( + await selectedRuns('--all', '--force', '--shard-count', '4', '--shard-index', shardIndex), + ).toEqual(unsharded); + } + }); +}); + +describe('flag validation', () => { + it('refuses to run without a run selector', async () => { + process.argv = ['bun', 'src/backfill-gpu-metrics.ts', '--dry-run']; + await expect(main()).rejects.toThrow('Pass --run or --all'); + expect(listedRunIds).toEqual([]); + }); + + it.each([ + { args: ['--all', '--since', '09/01/2026'], message: '--since requires a YYYY-MM-DD date' }, + { args: ['--all', '--parallel', '0'], message: '--parallel requires a positive integer' }, + { args: ['--run', '0'], message: '--run requires a positive integer' }, + ])('rejects $args', async ({ args, message }) => { + process.argv = ['bun', 'src/backfill-gpu-metrics.ts', '--dry-run', ...args]; + await expect(main()).rejects.toThrow(message); + expect(listedRunIds).toEqual([]); + }); +}); diff --git a/packages/db/src/lib/benchmark-result-lookup.test.ts b/packages/db/src/lib/benchmark-result-lookup.test.ts new file mode 100644 index 000000000..985a5723a --- /dev/null +++ b/packages/db/src/lib/benchmark-result-lookup.test.ts @@ -0,0 +1,219 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { PGlite } from '@electric-sql/pglite'; +import type postgres from 'postgres'; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from 'vitest'; + +import { mapBenchmarkRow, type BenchmarkParams } from '../etl/benchmark-mapper'; +import { createSkipTracker } from '../etl/skip-tracker'; +import { findBenchmarkResultIds, readMappedBenchmarkRows } from './benchmark-result-lookup'; + +type Sql = postgres.Sql; +let db: PGlite; +let sql: Sql; +const roots: string[] = []; + +function queryClient(database: Pick) { + const client = async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await database.query(query, values); + return result.rows; + }; + return Object.assign(client, { json: JSON.stringify, array: (value: unknown) => value }); +} + +const GITHUB_RUN_ID = 34557177019; +const RUN = { github_run_id: GITHUB_RUN_ID, run_attempt: 2 }; + +/** Raw `bmk_` JSON as the runner emits it for a fixed-sequence sweep. */ +function rawRow(overrides: Record = {}): Record { + return { + infmax_model_prefix: 'dsr1', + hw: 'b200-nv', + framework: 'sglang', + precision: 'fp4', + isl: 8192, + osl: 1024, + conc: 32, + tp: 8, + ep: 1, + dp_attention: false, + tput_per_gpu: 1234.5, + ...overrides, + }; +} + +function mapped(raw: Record): BenchmarkParams { + const row = mapBenchmarkRow(raw, createSkipTracker()); + if (!row) throw new Error('fixture did not map'); + return row; +} + +beforeAll(async () => { + db = await PGlite.create(); + const dir = new URL('../../migrations/', import.meta.url); + for (const name of fs.readdirSync(dir).toSorted()) { + if (name.endsWith('.sql')) await db.exec(fs.readFileSync(new URL(name, dir), 'utf8')); + } + sql = queryClient(db) as unknown as Sql; +}, 30_000); + +afterEach(() => { + for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); +}); + +afterAll(async () => { + await db?.close(); +}); + +/** + * Config 1 matches the fixture rows; config 2 differs only in decode_tp. Attempt 1 + * of the same GitHub run holds a point that must never be matched from attempt 2. + */ +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, created_at, date) + VALUES (1, ${GITHUB_RUN_ID}, 1, 'Run Sweep', '2026-09-11T04:00:00Z', '2026-09-11'), + (2, ${GITHUB_RUN_ID}, 2, 'Run Sweep', '2026-09-11T06:00:00Z', '2026-09-11'); + + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + is_multinode, prefill_tp, prefill_ep, prefill_dp_attention, prefill_num_workers, + decode_tp, decode_ep, decode_dp_attention, decode_num_workers, + num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, false, 8, 1, false, 0, + 8, 1, false, 0, 8, 8), + (2, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, false, 8, 1, false, 0, + 4, 1, false, 0, 8, 8), + (3, 'dsv4', 'b200', 'vllm', 'fp4', 'none', false, false, 1, 1, false, 0, + 1, 1, false, 0, 1, 1); + + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, + isl, osl, conc, offload_mode, metrics) + VALUES (20, 2, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, 'off', '{}'), + (21, 2, 1, 'single_turn', '2026-09-11', 8192, 1024, 64, 'off', '{}'), + (22, 2, 2, 'single_turn', '2026-09-11', 8192, 1024, 32, 'off', '{}'), + (23, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, 'off', '{}'), + (24, 2, 3, 'agentic_traces', '2026-09-11', null, null, 72, 'off', '{}');`); +}); + +describe('findBenchmarkResultIds', () => { + it('resolves one persisted point per mapped row using the run attempt and config key', async () => { + const ids = await findBenchmarkResultIds(sql, RUN, [ + mapped(rawRow()), + mapped(rawRow({ conc: 64 })), + ]); + expect(ids.toSorted()).toEqual([20, 21]); + }); + + it('never matches another attempt of the same run or a config with different parallelism', async () => { + // Point 23 is the same natural key under attempt 1; point 22 is attempt 2 but + // decode_tp 4. Neither may be linked to the attempt-2 telemetry. + expect( + await findBenchmarkResultIds(sql, { ...RUN, run_attempt: 3 }, [mapped(rawRow())]), + ).toEqual([]); + expect(await findBenchmarkResultIds(sql, RUN, [mapped(rawRow({ tp: 4 }))])).toEqual([]); + const attempt1 = await findBenchmarkResultIds(sql, { ...RUN, run_attempt: 1 }, [ + mapped(rawRow()), + ]); + expect(attempt1).toEqual([23]); + }); + + it('matches an agentic row whose isl and osl are null', async () => { + const agentic = mapped({ + infmax_model_prefix: 'dsv4', + hw: 'b200-nv', + framework: 'vllm', + precision: 'fp4', + scenario_type: 'agentic-coding', + users: 72, + tput_per_gpu: 20000, + }); + expect(agentic.isl).toBeNull(); + expect(await findBenchmarkResultIds(sql, RUN, [agentic])).toEqual([24]); + }); + + it('prefers the offload-exact point and reports no unique fallback', async () => { + await db.exec(`INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, + date, isl, osl, conc, offload_mode, metrics) + VALUES (25, 2, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, 'on', '{}');`); + const fallbacks: number[] = []; + const ids = await findBenchmarkResultIds(sql, RUN, [mapped(rawRow())], (id) => + fallbacks.push(id), + ); + expect(ids).toEqual([20]); + expect(fallbacks).toEqual([]); + }); + + it('accepts a lone offload-drifted point through the unique fallback', async () => { + // Historical mapper drift: the only surviving point carries 'on' while the + // current mapper derives 'off'. Attaching telemetry is still unambiguous. + await db.exec(`UPDATE benchmark_results SET offload_mode = 'on' WHERE id = 20;`); + const fallbacks: number[] = []; + const ids = await findBenchmarkResultIds(sql, RUN, [mapped(rawRow())], (id) => + fallbacks.push(id), + ); + expect(ids).toEqual([20]); + expect(fallbacks).toEqual([20]); + }); + + it('leaves ambiguous offload-drifted points unlinked', async () => { + await db.exec(`UPDATE benchmark_results SET offload_mode = 'on' WHERE id = 20; + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, + date, isl, osl, conc, offload_mode, metrics) + VALUES (26, 2, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, 'dram', '{}');`); + const fallbacks: number[] = []; + const ids = await findBenchmarkResultIds(sql, RUN, [mapped(rawRow())], (id) => + fallbacks.push(id), + ); + expect(ids).toEqual([]); + expect(fallbacks).toEqual([]); + }); +}); + +describe('readMappedBenchmarkRows', () => { + function writeArtifact(files: Record): string { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'bmk-lookup-')); + roots.push(root); + for (const [name, body] of Object.entries(files)) { + const target = path.join(root, name); + fs.mkdirSync(path.dirname(target), { recursive: true }); + fs.writeFileSync(target, body); + } + return root; + } + + it('maps every benchmark JSON under the artifact, accepting arrays and single objects', () => { + const root = writeArtifact({ + 'conc32.json': JSON.stringify(rawRow()), + 'nested/conc64.json': JSON.stringify([rawRow({ conc: 64 }), rawRow({ conc: 128 })]), + }); + const rows = readMappedBenchmarkRows(root); + expect(rows.map((row) => row.conc).toSorted((a, b) => a - b)).toEqual([32, 64, 128]); + expect(rows[0]?.config).toMatchObject({ + model: 'dsr1', + hardware: 'b200', + framework: 'sglang', + precision: 'fp4', + decodeTp: 8, + }); + }); + + it('drops rows the production mapper rejects and files that are not JSON', () => { + const root = writeArtifact({ + 'conc32.json': JSON.stringify([ + rawRow(), + rawRow({ hw: 'not-a-gpu' }), + rawRow({ conc: undefined }), + 'not-an-object', + null, + ]), + 'gpu_metrics.csv': 'timestamp, index\n2026/09/11 04:19:41.982, 0\n', + 'README.md': '# ignored', + }); + const rows = readMappedBenchmarkRows(root); + expect(rows).toHaveLength(1); + expect(rows[0]).toMatchObject({ conc: 32, isl: 8192, osl: 1024, offloadMode: 'off' }); + }); +}); diff --git a/packages/db/src/queries/gpu-metrics.test.ts b/packages/db/src/queries/gpu-metrics.test.ts new file mode 100644 index 000000000..23a90485b --- /dev/null +++ b/packages/db/src/queries/gpu-metrics.test.ts @@ -0,0 +1,295 @@ +import fs from 'node:fs'; + +import { PGlite } from '@electric-sql/pglite'; +import { afterAll, beforeAll, beforeEach, describe, expect, it } from 'vitest'; + +import type { DbClient } from '../connection'; +import { + getGpuMetricsAvailability, + getGpuMetricsForPoint, + getGpuMetricsForRun, +} from './gpu-metrics'; + +let db: PGlite; +const sql: DbClient = async (strings, ...values) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await db.query>(query, values); + return result.rows; +}; + +/** Stand-in client that fails loudly if a query slips past an early return. */ +const exploding: DbClient = () => { + throw new Error('should not query'); +}; + +const WITH_SERIES = 34557177019; +const WITHOUT_SERIES = 34557177020; +const RETRIED = 34557177021; + +beforeAll(async () => { + db = await PGlite.create(); + for (const name of ['001_initial_schema.sql', '016_gpu_metrics.sql']) { + await db.exec(fs.readFileSync(new URL(`../../migrations/${name}`, import.meta.url), 'utf8')); + } +}, 20_000); + +afterAll(async () => { + await db?.close(); +}); + +/** + * Run 1 holds a two-node artifact (one CSV per serving node) whose first CSV is + * shared by points 10 and 11; run 3 is a rerun of run 2's GitHub id. Point 12 + * is deliberately left unlinked. + */ +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, status, conclusion, + head_branch, head_sha, html_url, created_at, date) + VALUES (1, ${WITH_SERIES}, 1, 'Run Sweep', 'completed', 'success', 'main', 'abc123', + 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${WITH_SERIES}', + '2026-09-11T04:19:00Z', '2026-09-11'), + (2, ${WITHOUT_SERIES}, 1, 'Run Sweep', 'completed', 'success', null, null, null, + '2026-09-12T04:19:00Z', '2026-09-12'), + (3, ${RETRIED}, 1, 'Run Sweep', 'completed', 'failure', null, null, null, + '2026-09-13T04:19:00Z', '2026-09-13'), + (4, ${RETRIED}, 2, 'Run Sweep', 'completed', 'success', null, null, null, + '2026-09-13T06:19:00Z', '2026-09-13'); + + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + is_multinode, prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, true, 8, 8, 8, 8); + + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, + isl, osl, conc, metrics) + VALUES (10, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, '{}'), + (11, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 64, '{}'), + (12, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 128, '{}'); + + INSERT INTO gpu_metric_series (id, workflow_run_id, artifact_name, config_key, file_name, + vendor, csv_sha256, sample_interval_s, sample_count, gpu_count, started_at, ended_at, + sidecars) + VALUES (100, 1, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', 'node0/gpu_metrics.csv', 'nvidia', 'sha-node0', + 1, 3, 2, '2026-09-11T04:19:41Z', '2026-09-11T04:19:43Z', + '{"context": {"timestamp_timezone": "UTC"}}'), + (101, 1, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', 'node1/gpu_metrics.csv', 'nvidia', 'sha-node1', + 1, 1, 1, '2026-09-11T04:19:41Z', '2026-09-11T04:19:42Z', '{}'), + (102, 4, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_1', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_1', 'gpu_metrics.csv', 'amd', 'sha-attempt2', + null, 1, 1, '2026-09-13T06:20:00Z', '2026-09-13T06:20:01Z', '{}'), + (103, 3, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', 'gpu_metrics.csv', 'nvidia', 'sha-attempt1', + null, 1, 1, '2026-09-13T04:20:00Z', '2026-09-13T04:20:01Z', '{}'); + + INSERT INTO benchmark_result_gpu_metrics (benchmark_result_id, series_id) + VALUES (10, 100), (11, 100), (10, 101); + + INSERT INTO gpu_metric_gpu_stats (series_id, gpu_index, metric, sample_count, + min_value, max_value, mean_value, median_value, p95_value, p99_value, stddev_value) + VALUES (100, 0, 'powerW', 2, 187.5, 912.25, 549.875, 549.875, 912.25, 912.25, 362.375), + (100, 1, 'powerW', 1, 190.5, 190.5, 190.5, 190.5, 190.5, 190.5, 0), + (101, 0, 'powerW', 1, 500.5, 500.5, 500.5, 500.5, 500.5, 500.5, 0), + (102, 0, 'powerW', 1, 700.5, 700.5, 700.5, 700.5, 700.5, 700.5, 0), + (103, 0, 'powerW', 1, 100.5, 100.5, 100.5, 100.5, 100.5, 100.5, 0); + + INSERT INTO gpu_metric_samples (series_id, gpu_index, sampled_at, power_w, temperature_c, + sm_clock_mhz, mem_clock_mhz, gpu_util_pct, mem_util_pct, edge_temp_c, mem_temp_c, + gfx_voltage_mv, soc_voltage_mv, mem_voltage_mv, fclk_mhz, socclk_mhz, mm_activity_pct) + VALUES (100, 0, '2026-09-11T04:19:41Z', 187.5, 33, 120, 3996, 0, 0, + 35.5, 40.5, 750, 800, 1350, 1940, 1100, 12.5), + (100, 1, '2026-09-11T04:19:41Z', 190.5, 39, 120, 3996, 0, 0, + null, null, null, null, null, null, null, null), + -- Collector dropout: the row exists but every metric column is null. + (100, 0, '2026-09-11T04:19:42Z', null, null, null, null, null, null, + null, null, null, null, null, null, null, null), + (101, 0, '2026-09-11T04:19:41Z', 500.5, 50, 1965, 3996, 98, 74, + null, null, null, null, null, null, null, null), + (102, 0, '2026-09-13T06:20:00Z', 700.5, 60, 1965, 3996, 99, 80, + null, null, null, null, null, null, null, null), + (103, 0, '2026-09-13T04:20:00Z', 100.5, 20, 120, 3996, 0, 0, + null, null, null, null, null, null, null, null);`); +}); + +describe('getGpuMetricsForRun', () => { + it('returns null for a GitHub run id that was never ingested', async () => { + expect(await getGpuMetricsForRun(sql, 99999999999)).toBeNull(); + }); + + it('returns null for an ingested run that stored no telemetry series', async () => { + // The run row exists, so the null must come from the empty-series branch and + // not from a missing workflow_runs lookup. + const [row] = await sql`select id from workflow_runs where github_run_id = ${WITHOUT_SERIES}`; + expect(Number(row!.id)).toBe(2); + expect(await getGpuMetricsForRun(sql, WITHOUT_SERIES)).toBeNull(); + }); + + it('returns the run header, every series, its per-GPU stats, and its samples', async () => { + const payload = await getGpuMetricsForRun(sql, WITH_SERIES); + + expect(payload?.workflowRun).toEqual({ + id: 1, + githubRunId: WITH_SERIES, + runAttempt: 1, + name: 'Run Sweep', + date: '2026-09-11', + htmlUrl: `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${WITH_SERIES}`, + headBranch: 'main', + headSha: 'abc123', + conclusion: 'success', + status: 'completed', + createdAt: '2026-09-11T04:19:00.000Z', + }); + + expect(payload?.series.map((series) => series.fileName)).toEqual([ + 'node0/gpu_metrics.csv', + 'node1/gpu_metrics.csv', + ]); + + const [node0, node1] = payload!.series; + expect(node0).toMatchObject({ + id: 100, + artifactName: 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + configKey: 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 3, + gpuCount: 2, + startedAt: '2026-09-11T04:19:41.000Z', + endedAt: '2026-09-11T04:19:43.000Z', + sidecars: { context: { timestamp_timezone: 'UTC' } }, + benchmarkResultIds: [10, 11], + }); + expect(node0?.stats).toEqual([ + { + gpuIndex: 0, + metric: 'powerW', + count: 2, + min: 187.5, + max: 912.25, + mean: 549.875, + median: 549.875, + p95: 912.25, + p99: 912.25, + stddev: 362.375, + }, + { + gpuIndex: 1, + metric: 'powerW', + count: 1, + min: 190.5, + max: 190.5, + mean: 190.5, + median: 190.5, + p95: 190.5, + p99: 190.5, + stddev: 0, + }, + ]); + expect(node0?.data.map((sample) => [sample.timestamp, sample.index, sample.power])).toEqual([ + ['2026-09-11T04:19:41.000Z', 0, 187.5], + ['2026-09-11T04:19:41.000Z', 1, 190.5], + ['2026-09-11T04:19:42.000Z', 0, 0], + ]); + expect(node0?.data[0]).toEqual({ + timestamp: '2026-09-11T04:19:41.000Z', + index: 0, + power: 187.5, + temperature: 33, + smClock: 120, + memClock: 3996, + gpuUtil: 0, + memUtil: 0, + edgeTemp: 35.5, + memTemp: 40.5, + gfxVoltage: 750, + socVoltage: 800, + memVoltage: 1350, + fclk: 1940, + socClk: 1100, + mmActivity: 12.5, + }); + expect(node1?.benchmarkResultIds).toEqual([10]); + }); + + it('omits samples but keeps the digest when includeSamples is false', async () => { + const payload = await getGpuMetricsForRun(sql, WITH_SERIES, { includeSamples: false }); + expect(payload?.series.map((series) => series.data)).toEqual([[], []]); + expect(payload?.series[0]?.stats).toHaveLength(2); + // sampleCount is the stored column, so it still reports the full series length. + expect(payload?.series[0]?.sampleCount).toBe(3); + }); + + it('reads the latest attempt of a rerun GitHub run id, not the first', async () => { + const payload = await getGpuMetricsForRun(sql, RETRIED); + expect(payload?.workflowRun).toMatchObject({ id: 4, runAttempt: 2, conclusion: 'success' }); + expect(payload?.series.map((series) => ({ id: series.id, vendor: series.vendor }))).toEqual([ + { id: 102, vendor: 'amd' }, + ]); + }); + + it('reports a dropped sample as zeroes for the core metrics and drops the vendor extras', async () => { + const payload = await getGpuMetricsForRun(sql, WITH_SERIES); + const dropped = payload?.series[0]?.data.at(-1); + // Characterizes the `?? 0` defaulting: a null power reading is indistinguishable + // from a genuine 0 W reading once it reaches the chart. + expect(dropped).toEqual({ + timestamp: '2026-09-11T04:19:42.000Z', + index: 0, + power: 0, + temperature: 0, + smClock: 0, + memClock: 0, + gpuUtil: 0, + memUtil: 0, + edgeTemp: undefined, + memTemp: undefined, + gfxVoltage: undefined, + socVoltage: undefined, + memVoltage: undefined, + fclk: undefined, + socClk: undefined, + mmActivity: undefined, + }); + // The AMD-only columns are absent rather than zeroed, so consumers can tell + // "not collected" from "collected as zero" for those — but not for the core six. + expect(Object.hasOwn(dropped!, 'edgeTemp')).toBe(true); + expect(dropped?.edgeTemp).toBeUndefined(); + }); +}); + +describe('getGpuMetricsForPoint', () => { + it('returns null for a benchmark point with no linked series', async () => { + expect(await getGpuMetricsForPoint(sql, 12)).toBeNull(); + expect(await getGpuMetricsForPoint(sql, 9999)).toBeNull(); + }); + + it('returns every series linked to a multinode point, with each series full sample set', async () => { + const payload = await getGpuMetricsForPoint(sql, 10); + expect(payload?.benchmarkResultId).toBe(10); + expect(payload?.series.map((series) => [series.id, series.fileName])).toEqual([ + [100, 'node0/gpu_metrics.csv'], + [101, 'node1/gpu_metrics.csv'], + ]); + expect(payload?.series.map((series) => series.data.length)).toEqual([3, 1]); + // Each series carries every point that references it, not just the queried one. + expect(payload?.series.map((series) => series.benchmarkResultIds)).toEqual([[10, 11], [10]]); + + const single = await getGpuMetricsForPoint(sql, 11); + expect(single?.series.map((series) => series.id)).toEqual([100]); + }); +}); + +describe('getGpuMetricsAvailability', () => { + it('flags only the point ids that have at least one linked series', async () => { + expect(await getGpuMetricsAvailability(sql, [10, 11, 12, 9999])).toEqual({ + 10: true, + 11: true, + }); + }); + + it('short-circuits an empty id list without querying', async () => { + expect(await getGpuMetricsAvailability(exploding, [])).toEqual({}); + }); +}); From 855009f4e83e70efa558f5123cf67879b649c7b2 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 18 Sep 2026 15:02:33 -0700 Subject: [PATCH 010/103] fix: skip runs GitHub has deleted in artifact backfills MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `gh api …/runs//artifacts` returns HTTP 404 once a run is deleted or purged, and `retryArtifactOperation` spent the full backoff schedule on it before `backfill-gpu-metrics --all --dry-run` aborted on the first such run (34533943809, 18 runs into a 224-run sweep). Classify that 404 as `WorkflowRunNotFoundError`, a `NonRetryableArtifactError` the retry helper rethrows at once, and let both gpu-metrics and server-log backfills report the run as gone and continue. The gpu-metrics header also drops the wrong "GCS mirrors only bmk_/server_logs_" retention explanation. 中文:`gh api` 在 run 被删除或清理后返回 HTTP 404,之前 `retryArtifactOperation` 会把完整退避重试跑完,`backfill-gpu-metrics --all --dry-run` 在 224 个候选中 第 19 个 run(34533943809)处直接崩溃。现将该 404 归类为不可重试的 `WorkflowRunNotFoundError`,重试助手立即抛出,gpu-metrics 与 server-log 两个 backfill 记录该 run 已不存在并继续;同时修正 gpu-metrics 头注释中错误的 GCS 镜像保留说明。 --- packages/db/src/backfill-gpu-metrics.test.ts | 34 +++++++++++--- packages/db/src/backfill-gpu-metrics.ts | 47 ++++++++++++------- packages/db/src/backfill-server-log-files.ts | 16 ++++--- packages/db/src/lib/artifact-retry.test.ts | 20 +++++++- packages/db/src/lib/artifact-retry.ts | 13 ++++++ packages/db/src/lib/backfill-runner.ts | 25 ++++++++++ packages/db/src/lib/github-artifacts.test.ts | 49 +++++++++++++++++++- packages/db/src/lib/github-artifacts.ts | 46 ++++++++++++++++-- 8 files changed, 213 insertions(+), 37 deletions(-) diff --git a/packages/db/src/backfill-gpu-metrics.test.ts b/packages/db/src/backfill-gpu-metrics.test.ts index e718fdd4d..e54fe1d78 100644 --- a/packages/db/src/backfill-gpu-metrics.test.ts +++ b/packages/db/src/backfill-gpu-metrics.test.ts @@ -25,6 +25,8 @@ let db: PGlite; let sql: Sql; let main: () => Promise; const listedRunIds: string[] = []; +/** GitHub run ids whose artifact listing should 404 (the run was deleted). */ +const goneRunIds = new Set(); const originalArgv = process.argv; vi.mock('./etl/db-utils.js', async (importOriginal) => ({ @@ -32,13 +34,19 @@ vi.mock('./etl/db-utils.js', async (importOriginal) => ({ createAdminSql: () => sql, })); -vi.mock('./lib/github-artifacts.js', async (importOriginal) => ({ - ...(await importOriginal()), - listRunArtifacts: (_repository: string, runId: string): Promise => { - listedRunIds.push(runId); - return Promise.resolve([]); - }, -})); +vi.mock('./lib/github-artifacts.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRunArtifacts: (repository: string, runId: string): Promise => { + listedRunIds.push(runId); + if (goneRunIds.has(runId)) { + return Promise.reject(new actual.WorkflowRunNotFoundError(repository, runId)); + } + return Promise.resolve([]); + }, + }; +}); vi.mock('./lib/backfill-runner.js', async (importOriginal) => ({ ...(await importOriginal()), @@ -94,6 +102,7 @@ afterAll(async () => { */ beforeEach(async () => { listedRunIds.length = 0; + goneRunIds.clear(); vi.spyOn(console, 'log').mockImplementation(() => {}); await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, html_url, created_at, date) @@ -143,6 +152,17 @@ describe('candidate selection', () => { expect(await selectedRuns('--all')).toEqual([FRESH_NO_SERIES]); }); + it('a run GitHub has deleted is reported and skipped, not fatal for the sweep', async () => { + // GitHub 404s the artifact listing once a run is deleted (or purged past + // retention). One such run must not abort a months-long --all pass. + goneRunIds.add(String(FRESH_WITH_SERIES)); + expect(await selectedRuns('--all', '--force')).toEqual([FRESH_WITH_SERIES, FRESH_NO_SERIES]); + const lines = vi.mocked(console.log).mock.calls.map((call) => call.join(' ')); + expect(lines).toContainEqual(expect.stringContaining(`run ${FRESH_WITH_SERIES}`)); + expect(lines).toContainEqual(expect.stringContaining('gone from GitHub')); + expect(lines.at(-1)).toContain('1 gone from GitHub'); + }); + it('--run bypasses both the retention cutoff and the already-has-series filter', async () => { expect(await selectedRuns('--run', String(STALE_WITH_SERIES))).toEqual([STALE_WITH_SERIES]); }); diff --git a/packages/db/src/backfill-gpu-metrics.ts b/packages/db/src/backfill-gpu-metrics.ts index 5cbfa640d..5141256d8 100644 --- a/packages/db/src/backfill-gpu-metrics.ts +++ b/packages/db/src/backfill-gpu-metrics.ts @@ -3,9 +3,10 @@ * migration-016 tables for runs that were ingested before the CI path * digested them. * - * GitHub keeps run artifacts for 90 days and the GCS backup only mirrors - * bmk_/server_logs_ uploads, so the reachable history is bounded by GitHub - * retention. Each gpu_metrics artifact is paired with its exact `bmk_` + * GitHub keeps run artifacts for 90 days and the GCS mirror only covers + * scheduled/push runs on main, so the reachable history is bounded by GitHub + * retention; runs GitHub has since deleted are reported and skipped. Each + * gpu_metrics artifact is paired with its exact `bmk_` * (or `bmk_agentic_`) sibling, the raw rows are mapped through the * production mapper, and the series is linked to those persisted points. * @@ -29,9 +30,14 @@ import { AsyncSemaphore } from './etl/async-semaphore.js'; import { createAdminSql } from './etl/db-utils.js'; import { ingestGpuMetricsArtifact } from './etl/gpu-metrics-ingest.js'; import { retryArtifactOperation } from './lib/artifact-retry.js'; -import { confirmProceed, parseLimitForceFlags, runBackfillMain } from './lib/backfill-runner.js'; +import { + confirmProceed, + listBackfillRunArtifacts, + parseLimitForceFlags, + runBackfillMain, +} from './lib/backfill-runner.js'; import { findBenchmarkResultIds, readMappedBenchmarkRows } from './lib/benchmark-result-lookup.js'; -import { downloadArtifact, listRunArtifacts } from './lib/github-artifacts.js'; +import { downloadArtifact } from './lib/github-artifacts.js'; import { pairGpuMetricsArtifacts, type GpuMetricsArtifactPair, @@ -195,19 +201,23 @@ async function main(): Promise { if (flags.dryRun) { let pairedRuns = 0; let pairs = 0; + let goneRuns = 0; for (const run of runs) { const repository = repositoryFromRunUrl(run.html_url) ?? DEFAULT_REPO; - const artifacts = await retryArtifactOperation( - `listing GitHub artifacts for run ${run.github_run_id}`, - () => listRunArtifacts(repository, String(run.github_run_id)), - ); + const artifacts = await listBackfillRunArtifacts(repository, run.github_run_id); + if (artifacts === null) { + goneRuns++; + console.log(` run ${run.github_run_id} (${run.date}): gone from GitHub`); + continue; + } const runPairs = pairGpuMetricsArtifacts(artifacts); if (runPairs.length > 0) pairedRuns++; pairs += runPairs.length; console.log(` run ${run.github_run_id} (${run.date}): ${runPairs.length} pair(s)`); } console.log( - `\n=== dry run: ${runs.length} run(s), ${pairedRuns} with pairs, ${pairs} pair(s) ===`, + `\n=== dry run: ${runs.length} run(s), ${pairedRuns} with pairs, ${pairs} pair(s), ` + + `${goneRuns} gone from GitHub ===`, ); return; } @@ -225,6 +235,7 @@ async function main(): Promise { let artifactFailures = 0; let runFailures = 0; let missingRuns = 0; + let goneRuns = 0; for (const [runIndex, run] of runs.entries()) { const runId = run.github_run_id; @@ -232,10 +243,14 @@ async function main(): Promise { const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `gpu-metrics-backfill-${runId}-`)); const runStart = Date.now(); try { - const artifacts = await retryArtifactOperation( - `listing GitHub artifacts for run ${runId}`, - () => listRunArtifacts(repository, String(runId)), - ); + const artifacts = await listBackfillRunArtifacts(repository, runId); + if (artifacts === null) { + goneRuns++; + console.log( + ` [${runIndex + 1}/${runs.length}] run ${runId} attempt ${run.run_attempt}: gone from GitHub`, + ); + continue; + } const pairs = pairGpuMetricsArtifacts(artifacts); if (pairs.length === 0) { missingRuns++; @@ -293,8 +308,8 @@ async function main(): Promise { `\n=== backfill complete: ${artifactsProcessed} artifact(s), ${seriesStored} series, ` + `${samplesStored} sample(s), ${pointsLinked} point link(s), ` + `${unmatchedArtifacts} unmatched artifact(s), ${emptyArtifacts} empty artifact(s), ` + - `${missingRuns} run(s) without pairs, ${artifactFailures} failed artifact(s), ` + - `${runFailures} failed run(s) ===`, + `${missingRuns} run(s) without pairs, ${goneRuns} run(s) gone from GitHub, ` + + `${artifactFailures} failed artifact(s), ${runFailures} failed run(s) ===`, ); console.log(' Invalidate API cache after the backfill: bun run admin:cache:invalidate'); if (artifactFailures > 0 || runFailures > 0) process.exitCode = 1; diff --git a/packages/db/src/backfill-server-log-files.ts b/packages/db/src/backfill-server-log-files.ts index 3208194a2..46c0d05be 100644 --- a/packages/db/src/backfill-server-log-files.ts +++ b/packages/db/src/backfill-server-log-files.ts @@ -22,13 +22,18 @@ import { hasNoSslFlag } from './cli-utils.js'; import { insertServerLogFilePaths } from './etl/benchmark-ingest.js'; import { createAdminSql } from './etl/db-utils.js'; import { listServerLogFilePaths, serverLogArtifactRoot } from './etl/server-log-artifacts.js'; -import { downloadArtifact, listRunArtifacts } from './lib/github-artifacts.js'; +import { downloadArtifact } from './lib/github-artifacts.js'; import { downloadGcsArtifact, listGcsServerLogArtifacts, type GcsArtifactMeta, } from './lib/gcs-artifacts.js'; -import { confirmProceed, parseLimitForceFlags, runBackfillMain } from './lib/backfill-runner.js'; +import { + confirmProceed, + listBackfillRunArtifacts, + parseLimitForceFlags, + runBackfillMain, +} from './lib/backfill-runner.js'; import { retryArtifactOperation } from './lib/artifact-retry.js'; import { findBenchmarkResultIds, readMappedBenchmarkRows } from './lib/benchmark-result-lookup.js'; import { repositoryFromRunUrl } from './lib/runtime-metadata-artifacts.js'; @@ -197,11 +202,8 @@ async function main(): Promise { flags.source !== 'gcs' && (flags.source === 'github' || isWithinGithubRetention(run.date)) ) { - const artifacts = await retryArtifactOperation( - `listing GitHub artifacts for run ${runId}`, - () => listRunArtifacts(repository, String(runId)), - ); - pairs = pairServerLogArtifacts(artifacts.filter((artifact) => !artifact.expired)); + const artifacts = await listBackfillRunArtifacts(repository, runId); + pairs = pairServerLogArtifacts((artifacts ?? []).filter((artifact) => !artifact.expired)); if (pairs.length > 0) { source = 'github'; githubRuns++; diff --git a/packages/db/src/lib/artifact-retry.test.ts b/packages/db/src/lib/artifact-retry.test.ts index 675355b59..d369efc88 100644 --- a/packages/db/src/lib/artifact-retry.test.ts +++ b/packages/db/src/lib/artifact-retry.test.ts @@ -1,6 +1,6 @@ import { describe, expect, it, vi } from 'vitest'; -import { retryArtifactOperation } from './artifact-retry.js'; +import { NonRetryableArtifactError, retryArtifactOperation } from './artifact-retry.js'; describe('retryArtifactOperation', () => { it('retries transient failures using the configured delays', async () => { @@ -39,4 +39,22 @@ describe('retryArtifactOperation', () => { ).rejects.toBe(failure); expect(wait).toHaveBeenCalledOnce(); }); + + it('rethrows a non-retryable failure at once without waiting or warning', async () => { + // A deleted run 404s forever; sleeping through the full backoff schedule + // (~3.8 min at the default delays) per such run would stall a history sweep. + const gone = new NonRetryableArtifactError('run is gone'); + const operation = vi.fn(() => { + throw gone; + }); + const wait = vi.fn<(delayMs: number) => Promise>(() => Promise.resolve()); + const warn = vi.fn<(message: string) => void>(); + + await expect( + retryArtifactOperation('artifact', operation, { delaysMs: [5, 15], wait, warn }), + ).rejects.toBe(gone); + expect(operation).toHaveBeenCalledOnce(); + expect(wait).not.toHaveBeenCalled(); + expect(warn).not.toHaveBeenCalled(); + }); }); diff --git a/packages/db/src/lib/artifact-retry.ts b/packages/db/src/lib/artifact-retry.ts index e145aa04d..4044dc35b 100644 --- a/packages/db/src/lib/artifact-retry.ts +++ b/packages/db/src/lib/artifact-retry.ts @@ -1,5 +1,17 @@ const DEFAULT_RETRY_DELAYS_MS = [5_000, 15_000, 30_000, 60_000, 120_000] as const; +/** + * Marks a failure no retry can fix (the run or artifact is gone, not + * unreachable). `retryArtifactOperation` rethrows it immediately instead of + * burning the full backoff schedule on it. + */ +export class NonRetryableArtifactError extends Error { + constructor(message: string) { + super(message); + this.name = 'NonRetryableArtifactError'; + } +} + interface ArtifactRetryOptions { delaysMs?: readonly number[]; wait?: (delayMs: number) => Promise; @@ -26,6 +38,7 @@ export async function retryArtifactOperation( try { return await operation(); } catch (error) { + if (error instanceof NonRetryableArtifactError) throw error; lastError = error; if (attempt >= delaysMs.length) break; const delayMs = delaysMs[attempt]; diff --git a/packages/db/src/lib/backfill-runner.ts b/packages/db/src/lib/backfill-runner.ts index 876e559b2..2051e493d 100644 --- a/packages/db/src/lib/backfill-runner.ts +++ b/packages/db/src/lib/backfill-runner.ts @@ -8,6 +8,12 @@ import { confirm, hasYesFlag } from '../cli-utils.js'; import type { Sql } from '../etl/db-utils.js'; +import { retryArtifactOperation } from './artifact-retry.js'; +import { + WorkflowRunNotFoundError, + listRunArtifacts, + type ArtifactMeta, +} from './github-artifacts.js'; export interface LimitForceFlags { limit: number | null; @@ -165,6 +171,25 @@ export async function runCandidateIdBackfill( * class instances/prototypes so postgres.js serializes plain data only — * matches what the inline ingest path stores. */ +/** + * List a candidate run's GitHub artifacts with transient-failure retry. + * Returns `null` when GitHub no longer has the run at all, so a sweep over + * months of history reports the gap and moves on instead of aborting. + */ +export async function listBackfillRunArtifacts( + repository: string, + runId: number, +): Promise { + try { + return await retryArtifactOperation(`listing GitHub artifacts for run ${runId}`, () => + listRunArtifacts(repository, String(runId)), + ); + } catch (error) { + if (error instanceof WorkflowRunNotFoundError) return null; + throw error; + } +} + export function jsonbParam(sql: Sql, value: unknown): ReturnType { return sql.json(structuredClone(value) as unknown as Parameters[0]); } diff --git a/packages/db/src/lib/github-artifacts.test.ts b/packages/db/src/lib/github-artifacts.test.ts index 571643a5e..ae71f8890 100644 --- a/packages/db/src/lib/github-artifacts.test.ts +++ b/packages/db/src/lib/github-artifacts.test.ts @@ -1,6 +1,12 @@ import { describe, expect, it } from 'vitest'; -import { RUNNER_SUFFIX_RE, dedupeArtifactsByLogicalName } from './github-artifacts.js'; +import { NonRetryableArtifactError } from './artifact-retry.js'; +import { + RUNNER_SUFFIX_RE, + WorkflowRunNotFoundError, + dedupeArtifactsByLogicalName, + isGithubNotFoundError, +} from './github-artifacts.js'; const art = (name: string, created_at: string) => ({ name, @@ -40,3 +46,44 @@ describe('dedupeArtifactsByLogicalName', () => { expect(deduped.get('run-stats')?.name).toBe('run-stats'); }); }); + +describe('isGithubNotFoundError', () => { + it('recognises the status gh prints to stderr for a deleted run', () => { + // Shape of the error execSync throws with `encoding: 'utf8'`: the message + // only names the command; the HTTP status lives in stderr. + const error = Object.assign( + new Error('Command failed: gh api repos/o/r/actions/runs/1/artifacts'), + { + status: 1, + stderr: 'gh: Not Found (HTTP 404)\n{"message":"Not Found"}\n', + }, + ); + expect(isGithubNotFoundError(error)).toBe(true); + }); + + it('leaves transient failures retryable', () => { + expect( + isGithubNotFoundError( + Object.assign(new Error('Command failed'), { + stderr: 'gh: error connecting to api.github.com\n', + }), + ), + ).toBe(false); + expect( + isGithubNotFoundError( + Object.assign(new Error('Command failed'), { stderr: 'gh: Server Error (HTTP 502)\n' }), + ), + ).toBe(false); + expect(isGithubNotFoundError(null)).toBe(false); + expect(isGithubNotFoundError('HTTP 404')).toBe(false); + }); +}); + +describe('WorkflowRunNotFoundError', () => { + it('is non-retryable and names the run', () => { + const error = new WorkflowRunNotFoundError('SemiAnalysisAI/InferenceX', '34533943809'); + expect(error).toBeInstanceOf(NonRetryableArtifactError); + expect(error.runId).toBe('34533943809'); + expect(error.message).toContain('34533943809'); + }); +}); diff --git a/packages/db/src/lib/github-artifacts.ts b/packages/db/src/lib/github-artifacts.ts index 67827855e..c95957755 100644 --- a/packages/db/src/lib/github-artifacts.ts +++ b/packages/db/src/lib/github-artifacts.ts @@ -8,6 +8,8 @@ import { execSync } from 'node:child_process'; import fs from 'node:fs'; import path from 'node:path'; +import { NonRetryableArtifactError } from './artifact-retry.js'; + export interface ArtifactMeta { id?: number; name: string; @@ -32,12 +34,46 @@ export interface ArtifactMeta { */ export const RUNNER_SUFFIX_RE = /_[a-zA-Z][a-zA-Z0-9.-]*_\d+$/u; -/** List a workflow run's artifacts via `gh api` (paginated). Malformed lines are skipped. */ +/** + * GitHub no longer has the run (deleted, or purged past retention), so its + * artifact listing 404s. Distinct from expired artifacts, which still list + * with `expired: true`. + */ +export class WorkflowRunNotFoundError extends NonRetryableArtifactError { + readonly repo: string; + readonly runId: string; + + constructor(repo: string, runId: string) { + super(`GitHub has no run ${runId} in ${repo} (HTTP 404)`); + this.name = 'WorkflowRunNotFoundError'; + this.repo = repo; + this.runId = runId; + } +} + +/** `gh api` reports the HTTP status in its stderr (`gh: Not Found (HTTP 404)`). */ +export function isGithubNotFoundError(error: unknown): boolean { + if (typeof error !== 'object' || error === null) return false; + const { stderr, message } = error as { stderr?: unknown; message?: unknown }; + const text = [stderr, message].filter((part) => typeof part === 'string').join('\n'); + return /\(HTTP 404\)/u.test(text); +} + +/** + * List a workflow run's artifacts via `gh api` (paginated). Malformed lines + * are skipped; a 404 surfaces as `WorkflowRunNotFoundError`. + */ export function listRunArtifacts(repo: string, runId: string): ArtifactMeta[] { - const json = execSync( - `gh api "repos/${repo}/actions/runs/${runId}/artifacts" --paginate --jq '.artifacts[]'`, - { encoding: 'utf8', maxBuffer: 50 * 1024 * 1024 }, - ); + let json: string; + try { + json = execSync( + `gh api "repos/${repo}/actions/runs/${runId}/artifacts" --paginate --jq '.artifacts[]'`, + { encoding: 'utf8', maxBuffer: 50 * 1024 * 1024 }, + ); + } catch (error) { + if (isGithubNotFoundError(error)) throw new WorkflowRunNotFoundError(repo, runId); + throw error; + } const out: ArtifactMeta[] = []; for (const line of json.trim().split('\n')) { if (!line) continue; From 65c650101da6cc286985443b31ce3a66db1c9907 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 18 Sep 2026 15:49:33 -0700 Subject: [PATCH 011/103] feat: plan modeled power for aggregate multinode chassis at the deployment mean MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Aggregate multinode producers (Kimi K3 B200 dynamo-vLLM TP8/PP2, H200 vLLM TP16×2) emit no per-worker telemetry, so `modelSystemPower` rejected every such row as `topology` and the smart-provision basis showed no NVIDIA K3 SKU. Add a `uniform-hosts` topology basis for non-disaggregated multinode rows without a worker array: the telemetry GPU count must fill whole eight-GPU hosts and match TP×PP×PCP×replicas, and each chassis is modeled at the deployment mean; rows that carry per-worker telemetry keep the exact worker-hosts path and disaggregated rows still require it. `planningKwPerGpu` now admits every fully measured multi-chassis estimate instead of only one single-node chassis. Tooltip copy (en/zh) names the uniform-hosts assumption and the system-power doc records the admission rule. 中文:聚合多节点的采集端(Kimi K3 B200 dynamo-vLLM TP8/PP2、H200 vLLM TP16×2)不输出逐 worker 功耗,`modelSystemPower` 一律按 `topology` 拒绝, smart provision 里因此没有任何 NVIDIA K3 SKU。新增 `uniform-hosts` 拓扑基准: 非 disagg 多节点且无 worker 数组时,要求 telemetry GPU 数填满整数个八卡主机并 等于 TP×PP×PCP×副本数,每个机箱按部署平均功耗建模;带逐 worker 功耗的行仍走 精确的 worker-hosts 路径,disagg 行仍需逐 worker 数据。`planningKwPerGpu` 改为接受所有机箱均完整实测的估算。tooltip 中英文说明该假设,系统功耗文档同步。 --- docs/powerx-system-power.md | 11 ++- .../calculator/profit-power.test.ts | 38 +++++++++ .../src/components/calculator/profit-power.ts | 11 +-- .../inference/utils/tooltipUtils.ts | 5 +- .../app/src/lib/modeled-system-power.test.ts | 83 +++++++++++++++++++ packages/app/src/lib/modeled-system-power.ts | 40 ++++++++- 6 files changed, 175 insertions(+), 13 deletions(-) diff --git a/docs/powerx-system-power.md b/docs/powerx-system-power.md index efee20866..30a758ca2 100644 --- a/docs/powerx-system-power.md +++ b/docs/powerx-system-power.md @@ -95,9 +95,14 @@ facility kW/GPU used to calculate capacity per GW. Consequently, revenue, compute expense, license fee, and profit scale together; profit margin does not change. Electricity expense is not recomputed separately. -This opt-in AgentX estimate requires validated schema-v2 telemetry and a complete -single-node eight-GPU chassis supported by the pinned model. Partial allocations, -unsupported GB200/GB300 chassis, and missing/invalid measurements stay unavailable. +This opt-in AgentX estimate requires validated schema-v2 telemetry and fully +measured eight-GPU chassis supported by the pinned model: one single-node chassis, +one chassis per measured worker host, or, for an aggregate multinode deployment +whose producer emits no per-worker telemetry, every chassis at the deployment-mean +GPU power (`topologyBasis: 'uniform-hosts'`; symmetric TP/PP/DP shards load each +host alike). Partial allocations, unsupported GB200/GB300 chassis, disaggregated +deployments without per-worker telemetry, and missing/invalid measurements stay +unavailable. The ordinary 8K/1K transformation keeps its existing admission policy. At an exact frontier point, use that point's modeled power. Between points, diff --git a/packages/app/src/components/calculator/profit-power.test.ts b/packages/app/src/components/calculator/profit-power.test.ts index ce13042d8..37cc13173 100644 --- a/packages/app/src/components/calculator/profit-power.test.ts +++ b/packages/app/src/components/calculator/profit-power.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest'; import type { BenchmarkRow } from '@/lib/api'; import { modelSystemPower } from '@/lib/modeled-system-power'; +import { estimateChassisPower } from '@/lib/system-power-model'; import { Percentile, Sequence } from '@/lib/data-mappings'; import { buildGpuGroups, interpolateForGPU } from './useThroughputData'; import { estimateProfitRows } from './profit-estimator'; @@ -142,6 +143,43 @@ describe('profit power basis preview', () => { ).toBeNull(); }); + it('plans fully measured multi-chassis deployments at the same facility watts per GPU', () => { + // Kimi K3 B200 dynamo-vLLM TP8/PP2: sixteen GPUs on two hosts, aggregate + // producer without a per-worker array → uniform-hosts basis. + const multinode: BenchmarkRow = { + ...source, + hardware: 'b200', + framework: 'dynamo-vllm', + is_multinode: true, + num_prefill_gpu: 16, + num_decode_gpu: 16, + decode_num_workers: 1, + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 715.095, + avg_total_gpu_power_w: 11441.513, + decode_pp: 2, + }, + }; + expect(modelSystemPower(multinode, undefined, true)).toMatchObject({ + status: 'supported', + topologyBasis: 'uniform-hosts', + chassisBasis: 'full', + }); + const perChassis = estimateChassisPower('b200', 11441.513 / 2, 1.3)!; + expect( + modeledPowerAtTarget({ ...result, nearestPoints: [{ ...point, sourceRow: multinode }] }, 45), + ).toBeCloseTo(((perChassis.facilityWatts * 2) / 16 / 1000) * 1.1, 8); + // Disaggregated multinode rows still need per-worker telemetry. + expect( + modeledPowerAtTarget( + { ...result, nearestPoints: [{ ...point, sourceRow: { ...multinode, disagg: true } }] }, + 45, + ), + ).toBeNull(); + }); + it('leaves the default estimator and default AgentX model gate unchanged', () => { expect( estimateProfitByPower([result], specs, pricing, assumptions, 'provisioned', 45, labels), diff --git a/packages/app/src/components/calculator/profit-power.ts b/packages/app/src/components/calculator/profit-power.ts index 086322574..71308ed85 100644 --- a/packages/app/src/components/calculator/profit-power.ts +++ b/packages/app/src/components/calculator/profit-power.ts @@ -17,13 +17,10 @@ function planningKwPerGpu(point: GPUDataPoint): number | null { const row = point.sourceRow; if (!row || row.metrics.power_metric_schema_version !== 2) return null; const estimate = modelSystemPower(row, undefined, true); - if ( - estimate.status !== 'supported' || - estimate.gpuCount !== 8 || - estimate.topologyBasis !== 'single-node' || - estimate.chassisBasis !== 'full' - ) - return null; + // Every modeled chassis must be fully measured; a partial allocation's + // extrapolated share is not a planning figure. Multi-chassis deployments + // (per-worker or uniform hosts) plan at the same facility watts per GPU. + if (estimate.status !== 'supported' || estimate.chassisBasis !== 'full') return null; return (estimate.deploymentFacilityWatts / estimate.gpuCount / 1000) * 1.1; } diff --git a/packages/app/src/components/inference/utils/tooltipUtils.ts b/packages/app/src/components/inference/utils/tooltipUtils.ts index 314941a89..26626ed04 100644 --- a/packages/app/src/components/inference/utils/tooltipUtils.ts +++ b/packages/app/src/components/inference/utils/tooltipUtils.ts @@ -228,6 +228,8 @@ const SYSTEM_POWER_STRINGS = { : `${chassis} eight-GPU chassis · ${measured} of ${modeled} GPUs measured, extrapolated to full chassis`, extrapolation: 'Unmeasured chassis GPUs are assumed to run the same workload at the measured per-GPU power; deployment values are the measured GPUs’ share.', + uniformHosts: + 'No per-host telemetry for this multinode deployment; every chassis is modeled at the deployment-mean GPU power.', normalization: 'AC power is divided by all modeled chassis GPUs, including prefill and decode.', boundary: 'Includes GPU chassis CPUs; excludes separate CPU-only frontend/router hosts.', model: 'Power model source', @@ -257,6 +259,7 @@ const SYSTEM_POWER_STRINGS = { : `${chassis} 个八卡机箱 · 实测 ${measured}/${modeled} 张 GPU,按满机箱外推`, extrapolation: '假设机箱内未实测的 GPU 运行相同负载、功耗与实测每卡功耗相同;部署数值为实测 GPU 所占份额。', + uniformHosts: '该多节点部署没有逐主机功耗数据;每个机箱按部署平均每卡功耗建模。', normalization: '交流功耗按所有建模机箱的 GPU 总数分摊,包括 Prefill 与 Decode。', boundary: '计入 GPU 机箱内的 CPU;不计入独立的纯 CPU 前端或路由主机。', model: '功耗模型来源', @@ -303,7 +306,7 @@ const modeledSystemPowerHTML = ( ? ` ${tooltipLine(t.deploymentAc, `${fmt(estimate.deploymentAcWatts)} W`)} ${tooltipLine(`${t.facility} (PUE ${fmt(estimate.pue)})`, `${fmt(estimate.deploymentFacilityWatts)} W`)} -
${t.topology(estimate.chassisCount, estimate.gpuCount, estimate.modeledGpuCount)}${estimate.chassisBasis === 'extrapolated' ? `
${t.extrapolation}` : ''}
${t.assumptions}
${t.platformAssumptions}
${t.normalization}
${t.boundary}
+
${t.topology(estimate.chassisCount, estimate.gpuCount, estimate.modeledGpuCount)}${estimate.chassisBasis === 'extrapolated' ? `
${t.extrapolation}` : ''}${estimate.topologyBasis === 'uniform-hosts' ? `
${t.uniformHosts}` : ''}
${t.assumptions}
${t.platformAssumptions}
${t.normalization}
${t.boundary}
${tooltipLine(t.model, `${escapeHtml(estimate.hardware)} · ${escapeHtml(estimate.modelRevision.slice(0, 12))}`)} ${t.sweep} ` diff --git a/packages/app/src/lib/modeled-system-power.test.ts b/packages/app/src/lib/modeled-system-power.test.ts index c968537ae..c8fb4c26e 100644 --- a/packages/app/src/lib/modeled-system-power.test.ts +++ b/packages/app/src/lib/modeled-system-power.test.ts @@ -222,6 +222,89 @@ describe('modeled system power admission and accounting', () => { expect(modelSystemPower(source)).toMatchObject({ reason: 'role-power' }); }); + it('models an aggregate multinode deployment without worker telemetry at the deployment mean', () => { + // Kimi K3 B200 dynamo-vLLM TP8/PP2 (prod rows, 2026-09-17): two eight-GPU hosts, + // aggregate producer, no per-worker array. The K3 H200 vLLM row below is + // TP16 × 2 DP replicas across four hosts. + const b200 = row({ + is_multinode: true, + num_prefill_gpu: 16, + num_decode_gpu: 16, + decode_num_workers: 1, + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 715.095, + avg_total_gpu_power_w: 11441.513, + decode_pp: 2, + }, + }); + const perChassis = estimateChassisPower('b200', 11441.513 / 2, 1.3)!; + const estimate = modelSystemPower(b200); + expect(estimate).toMatchObject({ + status: 'supported', + topologyBasis: 'uniform-hosts', + chassisBasis: 'full', + gpuCount: 16, + chassisCount: 2, + modeledGpuCount: 16, + }); + if (estimate.status !== 'supported') throw new Error('unreachable'); + expect(estimate.chassisAcWatts).toBeCloseTo(perChassis.chassisAcWatts * 2, 6); + expect(estimate.deploymentFacilityWatts).toBe(estimate.facilityWatts); + expect(estimate.chassisAcWattsPerGpu).toBeCloseTo(perChassis.chassisAcWatts / 8, 6); + + const h200 = row({ + hardware: 'h200', + is_multinode: true, + prefill_tp: 16, + decode_tp: 16, + decode_ep: 32, + decode_dp_attention: true, + decode_num_workers: 2, + num_prefill_gpu: 32, + num_decode_gpu: 32, + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 167.357, + avg_total_gpu_power_w: 5355.413, + decode_pp: 1, + }, + }); + expect(modelSystemPower(h200)).toMatchObject({ + status: 'supported', + topologyBasis: 'uniform-hosts', + chassisCount: 4, + gpuCount: 32, + }); + + // A replica count that does not explain the telemetry width is not guessed around. + expect(modelSystemPower({ ...h200, decode_num_workers: 1 })).toMatchObject({ + reason: 'gpu-count', + }); + // Twelve GPUs cannot fill whole eight-GPU hosts; placement is unknown. + const twelve = row({ is_multinode: true, decode_tp: 12, prefill_tp: 12 }); + twelve.metrics.avg_total_gpu_power_w = twelve.metrics.avg_power_w * 12; + expect(modelSystemPower(twelve)).toMatchObject({ reason: 'topology' }); + // Per-worker telemetry, when present, keeps the more exact worker path. + const withWorkers = { + ...b200, + workers: ['host-a', 'host-b'].map((host, worker_idx) => ({ + role: 'agg', + worker_idx, + hosts: [host], + num_gpus: 8, + avg_power_w: 715.095, + })), + }; + expect(modelSystemPower(withWorkers)).toMatchObject({ + status: 'supported', + topologyBasis: 'worker-hosts', + chassisCount: 2, + }); + }); + it('preserves meaningful aggregate PP and PCP aliases before checking physical width', () => { for (const widths of [ { decode_pp: 1, prefill_pp: 2 }, diff --git a/packages/app/src/lib/modeled-system-power.ts b/packages/app/src/lib/modeled-system-power.ts index 550fe23fc..aadb737f6 100644 --- a/packages/app/src/lib/modeled-system-power.ts +++ b/packages/app/src/lib/modeled-system-power.ts @@ -43,7 +43,13 @@ export type SystemPowerEstimate = deploymentFacilityWatts: number; pue: number; telemetryBasis: 'validated-v2' | 'validated-unversioned-single-node'; - topologyBasis: 'single-node' | 'worker-hosts'; + /** + * 'single-node': one host, one chassis. 'worker-hosts': one chassis per + * measured worker, each at its own telemetry. 'uniform-hosts': an + * aggregate multinode deployment whose producer emitted no per-worker + * telemetry; every eight-GPU chassis is modeled at the deployment mean. + */ + topologyBasis: 'single-node' | 'worker-hosts' | 'uniform-hosts'; /** * 'full': every chassis had all eight GPUs measured. 'extrapolated': at least * one chassis was partially allocated; its model input is the measured per-GPU @@ -138,7 +144,7 @@ export function modelSystemPower( } const chassis: MeasuredChassis[] = []; - let topologyBasis: 'single-node' | 'worker-hosts'; + let topologyBasis: 'single-node' | 'worker-hosts' | 'uniform-hosts'; if (row.disagg === false && row.is_multinode === false) { // One host cannot hold more than one chassis. if (gpuCount > CHASSIS_GPU_COUNT) return unavailable('topology'); @@ -173,6 +179,36 @@ export function modelSystemPower( ? m.avg_total_gpu_power_w : m.avg_power_w * CHASSIS_GPU_COUNT, }); + } else if (row.disagg === false && (!Array.isArray(row.workers) || row.workers.length === 0)) { + // Aggregate multinode producers emit no per-worker telemetry. Symmetric + // TP/PP/DP shards load every host alike, so each full eight-GPU chassis is + // modeled at the deployment mean; the supported hardware only ships in + // eight-GPU hosts, so the count must fill whole chassis on several hosts. + // Disaggregated roles differ in load and stay on the worker path. + const hostCount = gpuCount / CHASSIS_GPU_COUNT; + if (!count(hostCount) || hostCount < 2) return unavailable('topology'); + const tp = row.decode_tp > 0 ? row.decode_tp : row.prefill_tp; + const pp = Math.max(m.pp ?? 1, m.decode_pp ?? 1, m.prefill_pp ?? 1); + const pcp = Math.max(m.pcp_size ?? 1, m.decode_pcp_size ?? 1, m.prefill_pcp_size ?? 1); + // Data-parallel replicas widen the deployment beyond one TP×PP×PCP group. + const replicas = Math.max(1, row.decode_num_workers); + if ( + !count(tp) || + !count(pp) || + !count(pcp) || + !count(replicas) || + tp * pp * pcp * replicas !== gpuCount + ) { + return unavailable('gpu-count'); + } + topologyBasis = 'uniform-hosts'; + for (let host = 0; host < hostCount; host++) { + chassis.push({ + measuredGpus: CHASSIS_GPU_COUNT, + // Partition the producer's exact total so the chassis inputs sum back to it. + modelInputWatts: m.avg_total_gpu_power_w / hostCount, + }); + } } else { // A role average across several hosts is insufficient for nonlinear // fan/PSU evaluation. Require one chassis per measured worker and a From 9ee4301545b4893202b008f440c8001f87e663d3 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Fri, 18 Sep 2026 18:06:31 -0700 Subject: [PATCH 012/103] feat: ingest multinode power_audit telemetry as per-host series MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Multinode InferenceX jobs upload no `gpu_metrics_` artifact; their per-GPU 1 Hz power lives in `power_audit_/LOGS/power/samples.csv`, one deployment-wide CSV from srt-slurm's `dcgm-power` collector. Every Kimi K3 NVIDIA point therefore had an empty PowerX tab while the data sat in GitHub. Accept the bundle as a fallback telemetry artifact: `gpuMetricsArtifactSuffix` recognises `power_audit_`, discovery (CI ingest) and backfill pairing prefer a `gpu_metrics_` sibling and use the bundle only when none exists, and `prepareGpuMetricsArtifact` regroups `samples.csv` by hostname into one power-only series per host (`file_name` `LOGS/power/samples.csv#`, manifest as context sidecar, GPU UUIDs as identity). Power-only series made a latent reader defect visible: `toSampleRow` coerced null clocks, temperature and utilization to 0, so the UI offered and plotted fabricated flat-zero metrics. The five non-power core fields are now optional end to end (reader, `GpuMetricRow`, anomaly checks, chart and correlation points, tooltip, correlation default). The PowerX tab labels the collector from the recorded producer instead of `nvidia-smi`, and takes the point's hardware key for the TDP line because multinode artifact names are hash-truncated. Docs record the adapter. Backfilled into the branch DB: K3 B200 34674595026 (7 points, 14 series), GB300 34873998796 (5 of 11; the six disaggregated jobs report `multinode_power_contract_missing`), H200 34744300699 (10 points, 40 series). Not covered here: the explorer's live GitHub fallback for not-yet-ingested runs still reads `gpu_metrics_` only. 中文:多节点 InferenceX 任务不上传 `gpu_metrics_` artifact,每 GPU 1 Hz 功耗数据 在 `power_audit_/LOGS/power/samples.csv`(srt-slurm `dcgm-power` 采集, 整个部署一个 CSV)里,Kimi K3 NVIDIA 数据点的 PowerX 标签页因此一直为空。 本次将该 bundle 作为回退 telemetry artifact:suffix 识别 `power_audit_`, CI 入库发现与 backfill 配对优先使用 `gpu_metrics_`,仅在没有时使用 bundle; `prepareGpuMetricsArtifact` 按 hostname 拆成每主机一条仅含功耗的序列。 同时修正读取端把空的时钟/温度/利用率读成 0 的问题(五个字段全链路改为可选), PowerX 标签页的 Collector 改为显示真实采集器,TDP 参考线改用数据点的硬件 key。 已回填 branch DB:K3 B200、GB300(5/11,disagg 任务无功耗合约)、H200。 未覆盖:explorer 对未入库 run 的 GitHub 实时回退仍只读 `gpu_metrics_`。 --- docs/data-pipeline.md | 12 +- .../gpu-power/GpuCorrelationChart.tsx | 7 +- .../components/gpu-power/GpuPowerChart.tsx | 30 ++++- .../components/gpu-power/GpuPowerDisplay.tsx | 13 +- .../src/components/gpu-power/types.test.ts | 44 +++++++ .../app/src/components/gpu-power/types.ts | 32 +++-- .../agentic-point/agentic-point-detail.tsx | 2 +- .../power-telemetry-view.test.ts | 26 ++++ .../agentic-point/power-telemetry-view.tsx | 21 +++- .../db/src/etl/gpu-metrics-artifacts.test.ts | 35 ++++++ packages/db/src/etl/gpu-metrics-artifacts.ts | 72 +++++++++-- .../db/src/etl/gpu-metrics-ingest.test.ts | 110 +++++++++++++++++ packages/db/src/etl/gpu-metrics-ingest.ts | 54 ++++++++- .../src/etl/multinode-power-samples.test.ts | 85 +++++++++++++ .../db/src/etl/multinode-power-samples.ts | 114 ++++++++++++++++++ packages/db/src/ingest-ci-run.ts | 7 +- .../db/src/lib/gpu-metrics-backfill.test.ts | 17 +++ packages/db/src/lib/gpu-metrics-backfill.ts | 15 ++- packages/db/src/queries/gpu-metrics.test.ts | 21 ++-- packages/db/src/queries/gpu-metrics.ts | 25 ++-- 20 files changed, 678 insertions(+), 64 deletions(-) create mode 100644 packages/app/src/components/inference/agentic-point/power-telemetry-view.test.ts create mode 100644 packages/db/src/etl/multinode-power-samples.test.ts create mode 100644 packages/db/src/etl/multinode-power-samples.ts diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index eaee38601..cf005cd11 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -529,9 +529,15 @@ Every single-node benchmark job (`benchmark-tmpl.yml`) samples `nvidia-smi` / `amd-smi` once per second for its whole lifetime and uploads the CSV as `gpu_metrics_` next to `bmk_` (agentic jobs: `bmk_agentic_`, still paired by the bare suffix). The multinode template uploads no `gpu_metrics_` -artifact — its telemetry travels only inside `power_audit_` — so multinode and -disaggregated points have no series here and their per-point PowerX tab stays -empty. The PowerX explorer used to download and parse the artifacts from GitHub +artifact; its telemetry travels inside `power_audit_` as +`LOGS/power/samples.csv`, one deployment-wide CSV written by srt-slurm's +`dcgm-power` collector (`timestamp_unix, hostname, gpu_index, gpu_uuid, power_w`, +power only). `etl/multinode-power-samples.ts` regroups it per host and the ingest +stores one series per host (`file_name` = `LOGS/power/samples.csv#`), +so multinode and disaggregated points get per-GPU power curves with null clocks, +temperature and utilization. Single-node jobs upload a `power_audit_` bundle too, +so discovery and backfill pairing use it only for a suffix with no `gpu_metrics_` +upload. The PowerX explorer used to download and parse the artifacts from GitHub on every request and lost them after GitHub's 90-day retention. CI ingest now digests them at ingest time, in the same step that links server logs: diff --git a/packages/app/src/components/gpu-power/GpuCorrelationChart.tsx b/packages/app/src/components/gpu-power/GpuCorrelationChart.tsx index 40b12b88e..fd8530eaf 100644 --- a/packages/app/src/components/gpu-power/GpuCorrelationChart.tsx +++ b/packages/app/src/components/gpu-power/GpuCorrelationChart.tsx @@ -68,7 +68,12 @@ const GpuCorrelationChart = React.memo( () => data .filter((r) => visibleGpus.has(r.index)) - .map((r) => ({ x: r[xMetric] ?? 0, y: r[yMetric] ?? 0, gpuIndex: r.index, raw: r })), + // Rows missing either metric were never sampled for it; skip them. + .flatMap((r) => { + const x = r[xMetric]; + const y = r[yMetric]; + return x === undefined || y === undefined ? [] : [{ x, y, gpuIndex: r.index, raw: r }]; + }), [data, visibleGpus, xMetric, yMetric], ); diff --git a/packages/app/src/components/gpu-power/GpuPowerChart.tsx b/packages/app/src/components/gpu-power/GpuPowerChart.tsx index bcfe74e72..addb25953 100644 --- a/packages/app/src/components/gpu-power/GpuPowerChart.tsx +++ b/packages/app/src/components/gpu-power/GpuPowerChart.tsx @@ -21,6 +21,7 @@ import { type GpuMetricRow, ALL_METRIC_OPTIONS, detectTdpFromArtifactName, + tdpForHardware, getGpuMetricLabel, getGpuMetricYAxisLabel, } from './types'; @@ -84,6 +85,12 @@ interface GpuMetricsChartProps { visibleGpus: Set; metricKey: GpuMetricKey; artifactName: string; + /** + * Hardware key of the benchmark point (`b200`, `h200`, …). Multinode bundle + * names are hash-truncated and carry no SKU token, so callers that know the + * point pass it explicitly; the artifact-name sniff stays the fallback. + */ + hardware?: string; legendElement?: React.ReactNode; caption?: React.ReactNode; /** Max interactive points before LTTB downsampling. Infinity to disable. */ @@ -123,11 +130,14 @@ function buildGroupedData( const groups = new Map(); for (const { row, ms } of parsed) { + const value = row[metricKey]; + // A metric the collector never sampled has no point, not a zero. + if (value === undefined) continue; if (!groups.has(row.index)) groups.set(row.index, []); groups.get(row.index)!.push({ seconds: (ms - minTime) / 1000, ms, - value: row[metricKey] ?? 0, + value, gpuIndex: row.index, raw: row, }); @@ -257,6 +267,7 @@ const GpuMetricsChart = React.memo( visibleGpus, metricKey, artifactName, + hardware, legendElement, caption, maxPoints, @@ -367,7 +378,10 @@ const GpuMetricsChart = React.memo( return ext; }, [allPoints]); - const tdpInfo = metricKey === 'power' ? detectTdpFromArtifactName(artifactName) : null; + const tdpInfo = + metricKey === 'power' + ? (tdpForHardware(hardware) ?? detectTdpFromArtifactName(artifactName)) + : null; const yDomain = useMemo(() => { if (allPoints.length === 0) return [0, 100] as [number, number]; @@ -510,9 +524,15 @@ const GpuMetricsChart = React.memo( ${rolling ? `
${t.rollingSuffix(display.windowS)}
` : ''} ${ d.raw - ? `
${t.power}${sep} ${d.raw.power.toFixed(1)} W
-
${t.temp}${sep} ${d.raw.temperature}\u00B0C
-
${t.utilization}${sep} ${d.raw.gpuUtil}%
` + ? `
${t.power}${sep} ${d.raw.power.toFixed(1)} W
${ + d.raw.temperature === undefined + ? '' + : `
${t.temp}${sep} ${d.raw.temperature}\u00B0C
` + }${ + d.raw.gpuUtil === undefined + ? '' + : `
${t.utilization}${sep} ${d.raw.gpuUtil}%
` + }` : '' } ${overlayRow} diff --git a/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx b/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx index 544795ed7..c5e5d4d77 100644 --- a/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx +++ b/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx @@ -241,6 +241,15 @@ export default function GpuMetricsDisplay() { const [chartView, setChartView] = useState('chart'); const [corrXMetric, setCorrXMetric] = useState('power'); const [corrYMetric, setCorrYMetric] = useState('temperature'); + // A power-only series (multinode DCGM bundle) has no temperature axis to + // default to; use the first other collected metric instead of an empty plot. + const effectiveCorrYMetric = useMemo( + () => + availableMetrics.some((m) => m.key === corrYMetric) + ? corrYMetric + : (availableMetrics.find((m) => m.key !== corrXMetric)?.key ?? corrXMetric), + [availableMetrics, corrXMetric, corrYMetric], + ); const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); const viewOptions = useMemo[]>( () => [ @@ -613,7 +622,7 @@ export default function GpuMetricsDisplay() {
handleMinInteractivityChange(e.target.value)} + onBlur={handleMinInteractivityBlur} + /> +
+ +
+ + handleCapsChange(e.target.value)} + onBlur={handleCapsBlur} + /> +
+ + {showsTcoBasis && ( +
+ + +
+ )} + + +
+ + + + {loading && ( + + + + )} + + {!loading && !hasAnyData && ( + +
+ {t.noData} +
+
+ )} + + {!loading && hasAnyData && ( + +
+ + {showsJalapenoPreview && } + {showsVeraRubinPreview && } + {showsTpuv7Preview && } + {hasBars ? ( + + ) : ( + <> +
{caption}
+
+ {result.measuredRows + result.overlayMeasuredRows === 0 + ? t.noTtft(statLabel) + : result.qualifyingRows + result.overlayQualifyingRows === 0 + ? t.noneQualify(minInteractivity, statLabel) + : t.noneUnderCaps(formatCap(Math.max(...caps)), statLabel)} +
+ + )} +
+ +

+ {t.note} + {t.methodology(statLabel, statLabel)} {t.source} + + {TCO_SOURCE_TITLE} + + +

+ + {tableRows.length > 0 && ( +
+ +
+ )} +
+ )} + {tcoModelDialog} +
+ ); +} diff --git a/packages/app/src/components/calculator/ProfitEstimatorDisplay.tsx b/packages/app/src/components/calculator/ProfitEstimatorDisplay.tsx index a3bef4ea9..a5367a87b 100644 --- a/packages/app/src/components/calculator/ProfitEstimatorDisplay.tsx +++ b/packages/app/src/components/calculator/ProfitEstimatorDisplay.tsx @@ -302,7 +302,7 @@ const STRINGS = { compareHistory: 'Compare history', gpuConfig: 'Chip Config', gpuConfigTooltip: `Select up to ${PROFIT_HISTORY_MAX_GPUS} chip configurations to compare how their estimated revenue and profit have moved over time. Each config is priced again on every compared date (the ends of the date range, plus any date or run added from the Config Changelog below) using the run measured then, so software updates show up as a change in the bar.`, - gpuConfigPlaceholder: 'Select a Chip Config for comparison', + gpuConfigPlaceholder: 'Select Chip Config', comparisonDateRange: 'Comparison Date Range', comparisonDateRangeTooltip: 'Select the start and end dates for the historical comparison. The chart adds a bar for each selected chip config at both dates, next to its bar for the run date shown above. Dates in between can be added one at a time from the Config Changelog.', @@ -416,7 +416,7 @@ const STRINGS = { compareHistory: '对比历史趋势', gpuConfig: '芯片配置', gpuConfigTooltip: `最多选择 ${PROFIT_HISTORY_MAX_GPUS} 个芯片配置,对比其收入与利润估算随时间的变化。每个配置都会用当日实测的运行结果,在每个对比日期(日期范围的起止两端,以及从下方配置变更日志中添加的日期或运行)重新估价,软件更新带来的差异会直接体现在柱形上。`, - gpuConfigPlaceholder: '选择芯片配置进行对比', + gpuConfigPlaceholder: '选择芯片配置', comparisonDateRange: '对比日期范围', comparisonDateRangeTooltip: '选择历史对比的起止日期。图表会在上方所示运行日期的柱形旁,为所选芯片配置在这两个日期各增加一根柱形。范围内的其他日期可在配置变更日志中逐个添加。', diff --git a/packages/app/src/components/calculator/cache-reuse.test.ts b/packages/app/src/components/calculator/cache-reuse.test.ts new file mode 100644 index 000000000..589d572a7 --- /dev/null +++ b/packages/app/src/components/calculator/cache-reuse.test.ts @@ -0,0 +1,288 @@ +import { describe, expect, it } from 'vitest'; + +import type { BenchmarkRow } from '@/lib/api'; + +import { + buildCacheReuse, + cacheShareOf, + defaultCacheReuseGroup, + formatShare, + tieredRowCount, +} from './cache-reuse'; +import type { GPUDataPoint } from './types'; + +function makeRow(overrides: Partial = {}): BenchmarkRow { + return { + id: 1, + hardware: 'b200', + framework: 'sglang', + model: 'glm5.2', + precision: 'fp4', + spec_method: 'none', + disagg: false, + is_multinode: false, + prefill_tp: 8, + decode_tp: 8, + num_prefill_gpu: 8, + num_decode_gpu: 8, + isl: null, + osl: null, + conc: 8, + offload_mode: 'on', + benchmark_type: 'agentic_traces', + image: 'sglang:test', + metrics: {}, + workers: null, + date: '2026-09-17', + run_url: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/1', + ...overrides, + } as BenchmarkRow; +} + +function makePoint( + metrics: Record, + row: Partial = {}, + point: Partial = {}, +): GPUDataPoint { + return { + hwKey: `${row.hardware ?? 'b200'}_${row.framework ?? 'sglang'}`, + interactivity: 100, + throughput: 1000, + outputThroughput: 300, + inputThroughput: 700, + concurrency: row.conc ?? 8, + tp: 8, + precision: 'fp4', + costh: 0.1, + costr: 0.2, + costhi: 0.05, + costri: 0.1, + costhOutput: 0.3, + costrOutput: 0.6, + tpPerMw: 0, + inputTpPerMw: 0, + outputTpPerMw: 0, + sourceRow: makeRow({ ...row, metrics }), + ...point, + } as GPUDataPoint; +} + +describe('cacheShareOf', () => { + it('reproduces the article: B200 + SGLang at concurrency 8, 12, 16', () => { + // Live GLM-5.2 rows, 2026-09-17. The article rounds the remainder to one decimal. + const rows = [ + [8, 0.903, 0.06, 0.056], + [12, 0.765, 0.192, 0.123], + [16, 0.548, 0.403, 0.368], + ] as const; + const shares = rows.map(([conc, gpu, cpu, external]) => + cacheShareOf( + makePoint( + { + server_gpu_cache_hit_rate: gpu, + server_cpu_cache_hit_rate: cpu, + server_external_cache_hit_rate: external, + }, + { conc }, + ), + ), + ); + // The CPU-offload figure is the HiCache host tier the article plots; the + // router's external figure is reported too and must not be added on top. + expect(shares.map((s) => s?.hostSource)).toEqual(['cpu', 'cpu', 'cpu']); + expect(shares.map((s) => formatShare(s!.hbm))).toEqual(['90.3%', '76.5%', '54.8%']); + expect(shares.map((s) => formatShare(s!.host))).toEqual(['6.0%', '19.2%', '40.3%']); + expect(shares.map((s) => formatShare(s!.unreused))).toEqual(['3.7%', '4.3%', '4.9%']); + }); + + it('falls back to the external tier when no CPU rate is reported', () => { + const share = cacheShareOf( + makePoint({ server_gpu_cache_hit_rate: 0.7, server_external_cache_hit_rate: 0.2 }), + ); + expect(share).toMatchObject({ hbm: 0.7, host: 0.2, hostSource: 'external', combined: false }); + expect(share!.unreused).toBeCloseTo(0.1); + }); + + it('draws TensorRT-LLM with offload as one combined reused segment', () => { + const share = cacheShareOf( + makePoint( + { server_gpu_cache_hit_rate: 0.85, server_cpu_cache_hit_rate: 0.3 }, + { framework: 'trt', offload_mode: 'on' }, + ), + ); + expect(share).toMatchObject({ hbm: 0.85, host: 0, combined: true, hostSource: null }); + expect(share!.unreused).toBeCloseTo(0.15); + }); + + it('keeps TensorRT-LLM without offload as a plain HBM figure', () => { + const share = cacheShareOf( + makePoint({ server_gpu_cache_hit_rate: 0.9 }, { framework: 'trt', offload_mode: 'off' }), + ); + expect(share).toMatchObject({ hbm: 0.9, host: 0, combined: false }); + }); + + it('reads the canonical offload descriptor over the legacy mode flag', () => { + const share = cacheShareOf( + makePoint( + { + server_gpu_cache_hit_rate: 0.8, + server_cpu_cache_hit_rate: 0.1, + // The descriptor is a string inside the numeric metrics bag on real rows. + kv_offloading: 'none' as unknown as number, + }, + { framework: 'trt', offload_mode: 'on' }, + ), + ); + expect(share!.combined).toBe(false); + }); + + it('clamps an over-reported runtime figure and keeps the raw total for the caveat', () => { + // GB300 Dynamo rows report 1.012 on occasion. + const share = cacheShareOf(makePoint({ server_gpu_cache_hit_rate: 1.012 })); + expect(share).toMatchObject({ hbm: 1, host: 0, unreused: 0 }); + expect(share!.reportedTotal).toBeCloseTo(1.012); + + const both = cacheShareOf( + makePoint({ server_gpu_cache_hit_rate: 0.9, server_external_cache_hit_rate: 0.3 }), + ); + expect(both!.host).toBeCloseTo(0.1); + expect(both!.unreused).toBe(0); + expect(both!.reportedTotal).toBeCloseTo(1.2); + }); + + it('carries the theoretical ceiling and returns null without any tier', () => { + expect( + cacheShareOf(makePoint({ server_gpu_cache_hit_rate: 0.5, theoretical_cache_hit_rate: 0.97 }))! + .theoretical, + ).toBe(0.97); + // A GB300 row scraped only at the frontend: ceiling but no measured tier. + expect(cacheShareOf(makePoint({ theoretical_cache_hit_rate: 0.97 }))).toBeNull(); + expect(cacheShareOf(makePoint({ tput_per_gpu: 100 }))).toBeNull(); + expect( + cacheShareOf({ ...makePoint({ server_gpu_cache_hit_rate: 0.5 }), sourceRow: undefined }), + ).toBeNull(); + }); +}); + +describe('buildCacheReuse', () => { + const official = [ + makePoint({ server_gpu_cache_hit_rate: 0.9, server_cpu_cache_hit_rate: 0.05 }, { conc: 8 }), + makePoint({ server_gpu_cache_hit_rate: 0.7, server_cpu_cache_hit_rate: 0.2 }, { conc: 16 }), + makePoint({ tput_per_gpu: 1 }, { conc: 32 }), + ]; + const config = { hwKey: 'b200_sglang' }; + + it('lists one official bar per concurrency and reports rows without tiers', () => { + const result = buildCacheReuse({ official, config }); + expect(result.series).toEqual([{ key: 'official', label: 'official' }]); + expect(result.concurrencies).toEqual([8, 16, 32]); + expect(result.bars.map((b) => [b.concurrency, b.share.hbm])).toEqual([ + [8, 0.9], + [16, 0.7], + ]); + expect(result.unmeasured).toMatchObject([{ seriesKey: 'official', concurrency: 32 }]); + expect(result.anyCombined).toBe(false); + }); + + it('adds each unofficial run on the same hardware as its own series', () => { + const result = buildCacheReuse({ + official, + config, + overlay: { + b200_sglang__run0: [ + makePoint( + { server_gpu_cache_hit_rate: 0.95, server_cpu_cache_hit_rate: 0.02 }, + { conc: 8 }, + ), + makePoint({ server_gpu_cache_hit_rate: 0.6 }, { conc: 24 }), + ], + // Another chip in the same run: not this configuration, so not drawn. + mi355x_sglang__run0: [ + makePoint({ server_gpu_cache_hit_rate: 0.5 }, { hardware: 'mi355x' }), + ], + b200_sglang__run1: [makePoint({ server_gpu_cache_hit_rate: 0.8 }, { conc: 8 })], + }, + overlayMeta: { + b200_sglang__run0: { hwKey: 'b200_sglang', runIndex: 0 }, + mi355x_sglang__run0: { hwKey: 'mi355x_sglang', runIndex: 0 }, + b200_sglang__run1: { hwKey: 'b200_sglang', runIndex: 1 }, + }, + overlayLabels: { 0: 'perf/hicache' }, + }); + expect(result.series.map((s) => s.key)).toEqual(['official', 'run:0', 'run:1']); + expect(result.series[1]).toMatchObject({ label: '✕ perf/hicache', runIndex: 0 }); + expect(result.series[2].label).toBe('✕ run 2'); + expect(result.concurrencies).toEqual([8, 16, 24, 32]); + expect(result.bars.filter((b) => b.seriesKey === 'run:0').map((b) => b.concurrency)).toEqual([ + 8, 24, + ]); + expect(result.bars.find((b) => b.seriesKey === 'run:0' && b.concurrency === 8)?.runIndex).toBe( + 0, + ); + }); + + it('gives an overlay-only configuration the full bar width', () => { + const result = buildCacheReuse({ + official: [], + config: { hwKey: 'mi355x_sglang' }, + overlay: { + mi355x_sglang__run0: [ + makePoint({ server_gpu_cache_hit_rate: 0.5 }, { hardware: 'mi355x', conc: 8 }), + ], + }, + overlayMeta: { mi355x_sglang__run0: { hwKey: 'mi355x_sglang', runIndex: 0 } }, + }); + expect(result.series.map((s) => s.key)).toEqual(['run:0']); + expect(result.bars.map((b) => b.seriesKey)).toEqual(['run:0']); + expect(result.unmeasured).toEqual([]); + }); + + it('matches overlay groups on precision when the page splits by precision', () => { + const result = buildCacheReuse({ + official, + config: { hwKey: 'b200_sglang', precision: 'fp4' }, + overlay: { + b200_sglang__fp8__run0: [makePoint({ server_gpu_cache_hit_rate: 0.5 }, { conc: 8 })], + b200_sglang__fp4__run0: [makePoint({ server_gpu_cache_hit_rate: 0.6 }, { conc: 8 })], + }, + overlayMeta: { + b200_sglang__fp8__run0: { hwKey: 'b200_sglang', precision: 'fp8', runIndex: 0 }, + b200_sglang__fp4__run0: { hwKey: 'b200_sglang', precision: 'fp4', runIndex: 0 }, + }, + }); + expect(result.bars.filter((b) => b.seriesKey === 'run:0')).toHaveLength(1); + expect(result.bars.find((b) => b.seriesKey === 'run:0')?.share.hbm).toBe(0.6); + }); +}); + +describe('defaultCacheReuseGroup', () => { + const tiered = (n: number, hardware: string, framework: string) => + Array.from({ length: n }, (_, i) => + makePoint({ server_gpu_cache_hit_rate: 0.5 }, { hardware, framework, conc: i + 1 }), + ); + + it('opens on the configuration with the most tiered rows', () => { + const groups = { + b300_sglang: tiered(3, 'b300', 'sglang'), + b200_vllm: tiered(5, 'b200', 'vllm'), + }; + const meta = { b300_sglang: { hwKey: 'b300_sglang' }, b200_vllm: { hwKey: 'b200_vllm' } }; + expect(defaultCacheReuseGroup(groups, meta)).toBe('b200_vllm'); + }); + + it('prefers SGLang on a tie and returns null when nothing reports a tier', () => { + const groups = { + b200_trt: tiered(4, 'b200', 'trt'), + b200_sglang: tiered(4, 'b200', 'sglang'), + h100_sglang: [makePoint({ tput_per_gpu: 1 }, { hardware: 'h100' })], + }; + const meta = { + b200_trt: { hwKey: 'b200_trt' }, + b200_sglang: { hwKey: 'b200_sglang' }, + h100_sglang: { hwKey: 'h100_sglang' }, + }; + expect(defaultCacheReuseGroup(groups, meta)).toBe('b200_sglang'); + expect(tieredRowCount(groups.h100_sglang)).toBe(0); + expect(defaultCacheReuseGroup({ h100_sglang: groups.h100_sglang }, meta)).toBeNull(); + }); +}); diff --git a/packages/app/src/components/calculator/cache-reuse.ts b/packages/app/src/components/calculator/cache-reuse.ts new file mode 100644 index 000000000..70537cfec --- /dev/null +++ b/packages/app/src/components/calculator/cache-reuse.ts @@ -0,0 +1,227 @@ +import { frameworkFamily } from '@/lib/framework-family'; +import { isKvOffloadEnabled } from '@/lib/kv-offload'; + +import type { GPUDataPoint } from './types'; +import type { GroupMeta, OverlayGroupMeta } from './useThroughputData'; + +/** + * Prefix-cache reuse: where a configuration's prompt tokens came from at each + * concurrency — HBM (the chip's own KV cache), the host tier behind it, or + * nowhere (recomputed) — as shares of all prompt tokens. + * + * Every bar is one measured row. The host segment is the CPU-offload rate + * (HiCache and its peers report host-memory hits there) and falls back to the + * router's external rate only when a row reports no CPU figure. This differs + * from `measuredCacheHitRate`, which prices cached input conservatively by + * preferring the external figure: that guards a revenue number against + * double-counting, whereas this chart names the tier a token was served from, + * and on SGLang rows the host figure is the CPU one. Both tiers are never + * summed here, so the bar cannot double count. + * + * TensorRT-LLM with offload enabled reports HBM and host as one combined figure + * (the point summary labels it "Combined chip + CPU"), so those rows draw one + * reused segment rather than inventing a split the data does not have. + */ + +export type CacheTier = 'hbm' | 'host' | 'unreused'; + +/** Tier colors, fixed rather than per-hardware: the bars compare tiers, not chips. */ +export const CACHE_TIER_COLORS: Record = { + hbm: '#3f9fe0', + host: '#f2b13a', + unreused: '#5f6672', +}; + +export interface CacheShare { + /** Prompt tokens served from the chip's KV cache, 0..1. */ + hbm: number; + /** Prompt tokens served from the host tier (CPU offload, else the router's external cache), 0..1. */ + host: number; + /** Prompt tokens recomputed, 0..1; the three shares sum to 1. */ + unreused: number; + /** The runtime folded host hits into the HBM figure, so `hbm` is both tiers. */ + combined: boolean; + hostSource: 'cpu' | 'external' | null; + /** Infinite-cache ceiling from the trace itself, when the row carries one. */ + theoretical: number | null; + /** Reported tiers summed before clamping; above 1 the runtime over-reported. */ + reportedTotal: number; +} + +const finite = (value: unknown): value is number => + typeof value === 'number' && Number.isFinite(value); +const clamp01 = (value: number) => Math.max(0, Math.min(1, value)); + +/** Prompt-token shares for one row, or null when the runtime reported no tier at all. */ +export function cacheShareOf(point: GPUDataPoint): CacheShare | null { + const row = point.sourceRow; + if (!row) return null; + const m = row.metrics; + const gpu = m.server_gpu_cache_hit_rate; + const external = m.server_external_cache_hit_rate; + const cpu = m.server_cpu_cache_hit_rate; + if (!finite(gpu) && !finite(external) && !finite(cpu)) return null; + + const offloadOn = isKvOffloadEnabled({ + kv_offloading: typeof m.kv_offloading === 'string' ? m.kv_offloading : null, + offload_mode: row.offload_mode, + }); + const combined = offloadOn && frameworkFamily(row.framework) === 'trt'; + + const hostSource: CacheShare['hostSource'] = combined + ? null + : finite(cpu) + ? 'cpu' + : finite(external) + ? 'external' + : null; + const hostRaw = hostSource === 'cpu' ? cpu! : hostSource === 'external' ? external! : 0; + const hbm = clamp01(finite(gpu) ? gpu : 0); + const host = Math.max(0, Math.min(1 - hbm, hostRaw)); + + return { + hbm, + host, + unreused: Math.max(0, 1 - hbm - host), + combined, + hostSource, + theoretical: finite(m.theoretical_cache_hit_rate) + ? clamp01(m.theoretical_cache_hit_rate) + : null, + reportedTotal: (finite(gpu) ? gpu : 0) + hostRaw, + }; +} + +export interface CacheReuseSeries { + /** `official`, or `run:` for an unofficial run. */ + key: string; + label: string; + runIndex?: number; +} + +export interface CacheReuseBar { + key: string; + concurrency: number; + seriesKey: string; + runIndex?: number; + point: GPUDataPoint; + share: CacheShare; +} + +export interface CacheReuseResult { + series: CacheReuseSeries[]; + /** Ascending, the union across series. */ + concurrencies: number[]; + bars: CacheReuseBar[]; + /** Rows of the chosen configuration that reported no cache tier, per series. */ + unmeasured: { seriesKey: string; concurrency: number; point: GPUDataPoint }[]; + anyCombined: boolean; +} + +export interface CacheReuseInput { + official: readonly GPUDataPoint[]; + /** Every overlay group; only those on the same hardware (and precision) are read. */ + overlay?: Record; + overlayMeta?: Record; + overlayLabels?: Record; + /** The chosen configuration, matched against overlay group metadata. */ + config: GroupMeta; +} + +/** + * The stacked bars for one configuration: its official rows, plus each loaded + * unofficial run's rows on the same hardware as their own series, so a branch + * can be read against the published curve concurrency by concurrency. + */ +export function buildCacheReuse(input: CacheReuseInput): CacheReuseResult { + // A configuration that exists only in a loaded run has no official slot; + // listing one anyway would halve every run bar beside an empty column. + const series: CacheReuseSeries[] = + input.official.length > 0 ? [{ key: 'official', label: 'official' }] : []; + const perSeries = new Map([['official', input.official]]); + + const runIndexes = new Set(); + for (const [groupKey, meta] of Object.entries(input.overlayMeta ?? {})) { + if (meta.hwKey !== input.config.hwKey) continue; + if (input.config.precision !== undefined && meta.precision !== input.config.precision) continue; + const points = input.overlay?.[groupKey]; + if (!points || points.length === 0) continue; + runIndexes.add(meta.runIndex); + const key = `run:${meta.runIndex}`; + perSeries.set(key, [...(perSeries.get(key) ?? []), ...points]); + } + for (const runIndex of [...runIndexes].toSorted((a, b) => a - b)) { + series.push({ + key: `run:${runIndex}`, + label: `✕ ${input.overlayLabels?.[runIndex] ?? `run ${runIndex + 1}`}`, + runIndex, + }); + } + + const bars: CacheReuseBar[] = []; + const unmeasured: CacheReuseResult['unmeasured'] = []; + const concurrencies = new Set(); + for (const entry of series) { + // One bar per concurrency: the latest official row is already unique per + // (config, concurrency); a run that repeats one keeps its first row. + const seen = new Set(); + for (const point of perSeries.get(entry.key) ?? []) { + if (seen.has(point.concurrency)) continue; + seen.add(point.concurrency); + concurrencies.add(point.concurrency); + const share = cacheShareOf(point); + if (!share) { + unmeasured.push({ seriesKey: entry.key, concurrency: point.concurrency, point }); + continue; + } + bars.push({ + key: `${point.concurrency}|${entry.key}`, + concurrency: point.concurrency, + seriesKey: entry.key, + ...(entry.runIndex === undefined ? {} : { runIndex: entry.runIndex }), + point, + share, + }); + } + } + + return { + series, + concurrencies: [...concurrencies].toSorted((a, b) => a - b), + bars, + unmeasured, + anyCombined: bars.some((b) => b.share.combined), + }; +} + +export function tieredRowCount(points: readonly GPUDataPoint[]): number { + return points.filter((p) => cacheShareOf(p) !== null).length; +} + +/** + * The configuration the page opens on: the one with the most rows reporting + * cache tiers, preferring an SGLang-family config on a tie because SGLang + * separates HBM from host hits. Null when no group reports any tier. + */ +export function defaultCacheReuseGroup( + groups: Record, + meta: Record, +): string | null { + let best: { key: string; tiered: number; sglang: boolean } | null = null; + for (const [key, points] of Object.entries(groups)) { + const tiered = tieredRowCount(points); + if (tiered === 0) continue; + const sglang = frameworkFamily(meta[key]?.hwKey ?? key) === 'sglang'; + if ( + best === null || + tiered > best.tiered || + (tiered === best.tiered && sglang && !best.sglang) || + (tiered === best.tiered && sglang === best.sglang && key.localeCompare(best.key) < 0) + ) { + best = { key, tiered, sglang }; + } + } + return best?.key ?? null; +} + +export const formatShare = (share: number): string => `${(share * 100).toFixed(1)}%`; diff --git a/packages/app/src/components/calculator/first-token-limits.test.ts b/packages/app/src/components/calculator/first-token-limits.test.ts new file mode 100644 index 000000000..71555e967 --- /dev/null +++ b/packages/app/src/components/calculator/first-token-limits.test.ts @@ -0,0 +1,324 @@ +import { describe, expect, it } from 'vitest'; + +import { + compareVendors, + DEFAULT_FIRST_TOKEN_CAPS, + formatFirstTokenCaps, + hardwareVendor, + MAX_FIRST_TOKEN_CAPS, + parseFirstTokenCaps, + selectFirstTokenWinners, + UNKNOWN_VENDOR, + ZH_MEDIAN, + zhStatPhrase, + formatCap, +} from './first-token-limits'; +import type { GPUDataPoint } from './types'; + +function makePoint(overrides: Partial = {}): GPUDataPoint { + return { + hwKey: 'b200_sglang', + interactivity: 160, + ttft: 1.5, + throughput: 1000, + outputThroughput: 300, + inputThroughput: 700, + concurrency: 8, + tp: 8, + precision: 'fp4', + costh: 0.1, + costr: 0.2, + costhi: 0.05, + costri: 0.1, + costhOutput: 0.3, + costrOutput: 0.6, + tpPerMw: 1000, + inputTpPerMw: 700, + outputTpPerMw: 300, + ...overrides, + }; +} + +const CAPS = [2, 5, 10]; + +/** The article's Figure 3 shape: B200 wins tight caps, GB300 wins looser ones. */ +const OFFICIAL = { + b200_sglang: [ + makePoint({ hwKey: 'b200_sglang', ttft: 1.6, costh: 0.067 }), + // Cheaper but too slow to start — must only appear under caps ≥ 5 s. + makePoint({ hwKey: 'b200_sglang', ttft: 4.2, costh: 0.06 }), + ], + 'gb300_dynamo-trt': [ + makePoint({ hwKey: 'gb300_dynamo-trt', ttft: 8.9, costh: 0.045 }), + // Fails the interactivity floor: never eligible, however cheap. + makePoint({ hwKey: 'gb300_dynamo-trt', ttft: 0.9, costh: 0.01, interactivity: 90 }), + ], + mi355x_sglang: [makePoint({ hwKey: 'mi355x_sglang', ttft: 1.4, costh: 0.127 })], +}; +const OFFICIAL_META = { + b200_sglang: { hwKey: 'b200_sglang' }, + 'gb300_dynamo-trt': { hwKey: 'gb300_dynamo-trt' }, + mi355x_sglang: { hwKey: 'mi355x_sglang' }, +}; + +const select = (extra: Partial[0]> = {}) => + selectFirstTokenWinners({ + official: OFFICIAL, + officialMeta: OFFICIAL_META, + caps: CAPS, + minInteractivity: 150, + costProvider: 'costh', + costType: 'total', + ...extra, + }); + +const winnerOf = ( + result: ReturnType, + cap: number, + seriesKey: string, +) => result.cells.find((c) => c.cap === cap && c.series.key === seriesKey)?.winner ?? null; + +describe('formatCap', () => { + it('prints caps the way the axis ticks do', () => { + expect([2, 0.5, 1 / 3].map(formatCap)).toEqual(['2', '0.5', '0.33']); + }); +}); + +describe('parseFirstTokenCaps', () => { + it('parses, de-duplicates, sorts, and drops junk', () => { + expect(parseFirstTokenCaps('10, 2,5,2,abc,-1,0')).toEqual([2, 5, 10]); + }); + + it('returns null when nothing is usable so callers fall back to the default', () => { + expect(parseFirstTokenCaps('')).toBeNull(); + expect(parseFirstTokenCaps(undefined)).toBeNull(); + expect(parseFirstTokenCaps('x,y')).toBeNull(); + }); + + it('caps the ladder length', () => { + const many = Array.from({ length: MAX_FIRST_TOKEN_CAPS + 4 }, (_, i) => i + 1).join(','); + expect(parseFirstTokenCaps(many)).toHaveLength(MAX_FIRST_TOKEN_CAPS); + }); + + it('round-trips the default ladder through the URL form', () => { + expect(parseFirstTokenCaps(formatFirstTokenCaps(DEFAULT_FIRST_TOKEN_CAPS))).toEqual([ + ...DEFAULT_FIRST_TOKEN_CAPS, + ]); + }); +}); + +describe('zhStatPhrase', () => { + it('puts the median after its noun and a percentile before it', () => { + expect(zhStatPhrase('交互性', ZH_MEDIAN)).toBe('交互性中位数'); + expect(zhStatPhrase('TTFT', ZH_MEDIAN)).toBe('TTFT 中位数'); + expect(zhStatPhrase('交互性', 'P90')).toBe('P90 交互性'); + }); + + it('spaces the phrase off the preceding text only when it starts with Latin', () => { + expect(zhStatPhrase('交互性', 'P90', '最低')).toBe('最低 P90 交互性'); + expect(zhStatPhrase('交互性', ZH_MEDIAN, '最低')).toBe('最低交互性中位数'); + }); +}); + +describe('hardwareVendor', () => { + it('reads the vendor off the registry base key, whatever the framework suffix', () => { + expect(hardwareVendor('b200_dynamo-sglang')).toBe('NVIDIA'); + expect(hardwareVendor('gb300-dynamo-trt')).toBe('NVIDIA'); + expect(hardwareVendor('mi355x_atom')).toBe('AMD'); + expect(hardwareVendor('tpuv7_vllm')).toBe('Google'); + }); + + it('falls back for hardware the registry does not know', () => { + expect(hardwareVendor('mystery_sglang')).toBe(UNKNOWN_VENDOR); + }); + + it('orders NVIDIA, then AMD, then everyone else alphabetically', () => { + expect(['Google', 'AMD', 'OpenAI', 'NVIDIA'].toSorted(compareVendors)).toEqual([ + 'NVIDIA', + 'AMD', + 'Google', + 'OpenAI', + ]); + }); +}); + +describe('selectFirstTokenWinners', () => { + it('lists one official series per vendor, NVIDIA first', () => { + expect(select().series.map((s) => s.key)).toEqual(['vendor:NVIDIA', 'vendor:AMD']); + }); + + it('picks the cheapest row under each cap that also clears the interactivity floor', () => { + const result = select(); + // Under 2 s only the fast B200 row qualifies for NVIDIA; the cheap GB300 + // row is too slow to start and the cheaper-still GB300 row is below 150 tok/s. + expect(winnerOf(result, 2, 'vendor:NVIDIA')).toMatchObject({ + hwKey: 'b200_sglang', + cost: 0.067, + }); + // At 5 s the slower-to-start, cheaper B200 row takes over. + expect(winnerOf(result, 5, 'vendor:NVIDIA')).toMatchObject({ + hwKey: 'b200_sglang', + cost: 0.06, + }); + // At 10 s GB300 finally clears the cap and is cheapest of all. + expect(winnerOf(result, 10, 'vendor:NVIDIA')).toMatchObject({ + hwKey: 'gb300_dynamo-trt', + cost: 0.045, + }); + for (const cap of CAPS) { + expect(winnerOf(result, cap, 'vendor:AMD')).toMatchObject({ + hwKey: 'mi355x_sglang', + cost: 0.127, + }); + } + }); + + it('summarizes the cross-vendor gap as a fraction of the runner-up', () => { + const [two, , ten] = select().summaries; + expect(two.best?.hwKey).toBe('b200_sglang'); + expect(two.runnerUp?.hwKey).toBe('mi355x_sglang'); + expect(two.pctLower).toBeCloseTo((0.127 - 0.067) / 0.127, 10); + expect(ten.best?.hwKey).toBe('gb300_dynamo-trt'); + expect(ten.pctLower).toBeCloseTo((0.127 - 0.045) / 0.127, 10); + }); + + it('leaves a vendor column empty rather than dropping the vendor when nothing qualifies', () => { + const result = select({ caps: [1] }); + expect(result.series.map((s) => s.key)).toEqual(['vendor:NVIDIA', 'vendor:AMD']); + expect(winnerOf(result, 1, 'vendor:NVIDIA')).toBeNull(); + expect(winnerOf(result, 1, 'vendor:AMD')).toBeNull(); + expect(result.summaries[0]).toMatchObject({ best: null, runnerUp: null, pctLower: null }); + }); + + it('has no gap when only one vendor qualifies', () => { + const result = select({ minInteractivity: 155, official: { ...OFFICIAL, mi355x_sglang: [] } }); + expect(result.summaries[0].best?.hwKey).toBe('b200_sglang'); + expect(result.summaries[0].runnerUp).toBeNull(); + expect(result.summaries[0].pctLower).toBeNull(); + }); + + it('respects the legend: hidden hardware is not a candidate', () => { + const result = select({ + visibleHwKeys: new Set(['gb300_dynamo-trt', 'mi355x_sglang']), + }); + expect(winnerOf(result, 2, 'vendor:NVIDIA')).toBeNull(); + expect(winnerOf(result, 10, 'vendor:NVIDIA')?.hwKey).toBe('gb300_dynamo-trt'); + }); + + it('reads the cost for the selected pricing tier and token type', () => { + const rental = select({ costProvider: 'costr' }); + expect(winnerOf(rental, 2, 'vendor:NVIDIA')?.cost).toBe(0.2); + const output = select({ costType: 'output' }); + expect(winnerOf(output, 2, 'vendor:AMD')?.cost).toBe(0.3); + }); + + it('ignores rows with no usable first-token measurement', () => { + const result = select({ + official: { + ...OFFICIAL, + b200_sglang: [ + makePoint({ hwKey: 'b200_sglang', ttft: undefined, costh: 0.001 }), + makePoint({ hwKey: 'b200_sglang', ttft: 0, costh: 0.002 }), + makePoint({ hwKey: 'b200_sglang', ttft: Number.NaN, costh: 0.003 }), + ], + }, + }); + for (const cap of CAPS) + expect(winnerOf(result, cap, 'vendor:NVIDIA')?.hwKey).toBe( + cap >= 10 ? 'gb300_dynamo-trt' : undefined, + ); + // The unreadable rows are not counted as measured either. + expect(result.measuredRows).toBe(3); + }); + + it('breaks cost ties toward the shorter first token, then the faster stream', () => { + const result = select({ + official: { + a: [ + makePoint({ hwKey: 'b200_sglang', ttft: 1.9, costh: 0.05, interactivity: 160 }), + makePoint({ hwKey: 'b200_sglang', ttft: 1.2, costh: 0.05, interactivity: 155 }), + makePoint({ hwKey: 'b200_sglang', ttft: 1.2, costh: 0.05, interactivity: 170 }), + ], + }, + officialMeta: { a: { hwKey: 'b200_sglang' } }, + }); + expect(winnerOf(result, 2, 'vendor:NVIDIA')).toMatchObject({ ttft: 1.2, interactivity: 170 }); + }); + + it('counts qualifying and measured rows for the subtitle', () => { + const result = select(); + expect(result.measuredRows).toBe(5); + // The 90 tok/s GB300 row fails the floor. + expect(result.qualifyingRows).toBe(4); + }); + + describe('unofficial-run overlays', () => { + const OVERLAY = { + b200_sglang__run0: [makePoint({ hwKey: 'b200_sglang', ttft: 1.1, costh: 0.03 })], + mi355x_sglang__run0: [makePoint({ hwKey: 'mi355x_sglang', ttft: 1, costh: 0.02 })], + b300_sglang__run1: [makePoint({ hwKey: 'b300_sglang', ttft: 7, costh: 0.01 })], + }; + const OVERLAY_META = { + b200_sglang__run0: { hwKey: 'b200_sglang', runIndex: 0 }, + mi355x_sglang__run0: { hwKey: 'mi355x_sglang', runIndex: 0 }, + b300_sglang__run1: { hwKey: 'b300_sglang', runIndex: 1 }, + }; + + it('adds one series per run after the vendors, labelled with the branch', () => { + const result = select({ + overlay: OVERLAY, + overlayMeta: OVERLAY_META, + overlayLabels: { 0: 'perf/faster-prefill' }, + }); + expect(result.series.map((s) => s.key)).toEqual([ + 'vendor:NVIDIA', + 'vendor:AMD', + 'run:0', + 'run:1', + ]); + expect(result.series[2]).toMatchObject({ label: '✕ perf/faster-prefill', runIndex: 0 }); + expect(result.series[3].label).toBe('✕ run 2'); + }); + + it('picks a run’s cheapest qualifying row across every vendor it touched', () => { + const result = select({ overlay: OVERLAY, overlayMeta: OVERLAY_META }); + expect(winnerOf(result, 2, 'run:0')).toMatchObject({ hwKey: 'mi355x_sglang', cost: 0.02 }); + expect(winnerOf(result, 2, 'run:1')).toBeNull(); + expect(winnerOf(result, 10, 'run:1')).toMatchObject({ hwKey: 'b300_sglang', runIndex: 1 }); + }); + + it('never lets an overlay row into the official vendor bars or the gap summary', () => { + const result = select({ overlay: OVERLAY, overlayMeta: OVERLAY_META }); + expect(winnerOf(result, 2, 'vendor:AMD')?.cost).toBe(0.127); + expect(result.summaries[0].best?.cost).toBe(0.067); + }); + + it('counts readable overlay rows separately so an overlay-only page is not "no TTFT"', () => { + const result = select({ + official: {}, + officialMeta: {}, + overlay: { + b200_sglang__run0: [ + makePoint({ hwKey: 'b200_sglang', ttft: 1, interactivity: 90 }), + makePoint({ hwKey: 'b200_sglang', ttft: 40, interactivity: 200 }), + ], + }, + overlayMeta: { b200_sglang__run0: { hwKey: 'b200_sglang', runIndex: 0 } }, + }); + expect(result.measuredRows).toBe(0); + expect(result.overlayMeasuredRows).toBe(2); + expect(result.overlayQualifyingRows).toBe(1); + expect(winnerOf(result, 2, 'run:0')).toBeNull(); + }); + + it('hides overlay rows for hardware the legend has switched off', () => { + const result = select({ + overlay: OVERLAY, + overlayMeta: OVERLAY_META, + visibleHwKeys: new Set(['b200_sglang', 'gb300_dynamo-trt']), + }); + expect(winnerOf(result, 2, 'run:0')).toMatchObject({ hwKey: 'b200_sglang', cost: 0.03 }); + expect(result.series.map((s) => s.key)).not.toContain('run:1'); + }); + }); +}); diff --git a/packages/app/src/components/calculator/first-token-limits.ts b/packages/app/src/components/calculator/first-token-limits.ts new file mode 100644 index 000000000..7b939b03a --- /dev/null +++ b/packages/app/src/components/calculator/first-token-limits.ts @@ -0,0 +1,330 @@ +import { HW_REGISTRY } from '@semianalysisai/inferencex-constants'; + +import { getCostField } from './interpolation'; +import type { CostProvider, CostType, GPUDataPoint } from './types'; +import type { GroupMeta, OverlayGroupMeta } from './useThroughputData'; + +/** + * First-token limits: the cheapest *measured* configuration per vendor that + * clears an interactivity floor while keeping time-to-first-token under a cap. + * + * This deliberately reads measured rows, not the interpolated Pareto frontier + * the calculator uses. A TTFT cap is a hard operating constraint, and the + * frontier only tracks throughput against interactivity — a knot that + * interpolates cheaply at the target may sit between two rows whose first-token + * waits are seconds apart. Reading rows keeps every bar a configuration that + * actually ran under both constraints at once, which is also what an article + * can cite: a run URL, not a spline. + * + * The interactivity and TTFT filters are separate percentile statistics on the + * same row. Passing both says the row's p90 streaming speed and its p90 first + * token both cleared their bars, not that every individual request did. + */ + +/** Default TTFT ladder, in seconds. */ +export const DEFAULT_FIRST_TOKEN_CAPS: readonly number[] = [2, 5, 10, 15, 20]; + +/** Hard ceiling on ladder length so a pasted URL cannot render hundreds of groups. */ +export const MAX_FIRST_TOKEN_CAPS = 8; + +/** + * Interactivity floor the page opens on. AgentX publishes its cost comparisons + * at 150 tok/s/user; fixed sequences are read at the calculator's default. + */ +export const DEFAULT_FIRST_TOKEN_MIN_INTERACTIVITY = { agentic: 150, fixed: 35 } as const; + +/** + * Parse a comma-separated ladder such as `2,5,10`. Values are de-duplicated, + * sorted ascending, and capped at {@link MAX_FIRST_TOKEN_CAPS}. Returns null + * when nothing usable remains, so callers can fall back to the default. + */ +export function parseFirstTokenCaps(raw: string | null | undefined): number[] | null { + if (!raw) return null; + const caps = new Set(); + for (const part of raw.split(',')) { + const value = Number.parseFloat(part.trim()); + if (Number.isFinite(value) && value > 0) caps.add(value); + } + if (caps.size === 0) return null; + return [...caps].toSorted((a, b) => a - b).slice(0, MAX_FIRST_TOKEN_CAPS); +} + +export function formatFirstTokenCaps(caps: readonly number[]): string { + return caps.join(','); +} + +/** Two decimals is as fine as a cap gets typed; `Number()` drops the zeros `toFixed` pads. */ +export function formatCap(cap: number): string { + return Number.isInteger(cap) ? String(cap) : String(Number(cap.toFixed(2))); +} + +/** Chinese label of the median statistic; percentiles stay `P90` / `P75`. */ +export const ZH_MEDIAN = '中位数'; + +/** + * Chinese names a statistic differently for the median and for a percentile: + * the median follows its noun (`交互性中位数`, `TTFT 中位数`) while a percentile + * precedes it (`P90 交互性`), the way the rest of the site writes them. `before` + * is the text the phrase continues; a space separates the two only when the + * phrase begins with Latin text, matching the site's CJK/Latin spacing. + */ +export function zhStatPhrase(noun: string, stat: string, before = ''): string { + const phrase = + stat === ZH_MEDIAN + ? `${noun}${/[A-Za-z]$/u.test(noun) ? ' ' : ''}${ZH_MEDIAN}` + : `${stat} ${noun}`; + const gap = before !== '' && /^[A-Za-z]/u.test(phrase) ? ' ' : ''; + return `${before}${gap}${phrase}`; +} + +/** Vendors shown first, in this order; any other vendor follows alphabetically. */ +const VENDOR_ORDER = ['NVIDIA', 'AMD']; + +export const UNKNOWN_VENDOR = 'Other'; + +/** Vendor for a hardware key — `b200_dynamo-sglang` → `NVIDIA`. */ +export function hardwareVendor(hwKey: string): string { + const base = hwKey.split(/[-_]/u)[0]; + return HW_REGISTRY[base]?.vendor ?? UNKNOWN_VENDOR; +} + +export function compareVendors(a: string, b: string): number { + const ia = VENDOR_ORDER.indexOf(a); + const ib = VENDOR_ORDER.indexOf(b); + if (ia !== -1 || ib !== -1) { + if (ia === -1) return 1; + if (ib === -1) return -1; + return ia - ib; + } + return a.localeCompare(b); +} + +/** One bar family: a vendor's official rows, or one unofficial run's rows. */ +export interface FirstTokenSeries { + /** `vendor:` for official series, `run:` for overlays. */ + key: string; + label: string; + vendor?: string; + runIndex?: number; +} + +/** The configuration a bar stands for. */ +export interface FirstTokenWinner { + seriesKey: string; + hwKey: string; + vendor: string; + precision: string; + /** $/M tokens at the selected pricing tier and token type. */ + cost: number; + /** Seconds, at the percentile the page reads. */ + ttft: number; + /** tok/s/user, at the percentile the page reads. */ + interactivity: number; + point: GPUDataPoint; + runIndex?: number; +} + +export interface FirstTokenCell { + cap: number; + series: FirstTokenSeries; + /** Null when nothing in the series cleared both constraints. */ + winner: FirstTokenWinner | null; +} + +/** How the official vendors compare under one cap. */ +export interface FirstTokenCapSummary { + cap: number; + best: FirstTokenWinner | null; + runnerUp: FirstTokenWinner | null; + /** `(runnerUp − best) / runnerUp`; null without two qualifying vendors. */ + pctLower: number | null; +} + +export interface FirstTokenResult { + series: FirstTokenSeries[]; + cells: FirstTokenCell[]; + summaries: FirstTokenCapSummary[]; + /** Official rows that cleared the interactivity floor, before any TTFT cap. */ + qualifyingRows: number; + /** Official rows the page could read at all: visible hardware with a TTFT. */ + measuredRows: number; + /** Same, for the loaded unofficial runs; kept apart so the caption stays official-only. */ + overlayMeasuredRows: number; + overlayQualifyingRows: number; +} + +export interface FirstTokenSelectionInput { + official: Record; + officialMeta: Record; + overlay?: Record; + overlayMeta?: Record; + /** Branch (or fallback) label per run index, for the overlay series label. */ + overlayLabels?: Record; + caps: readonly number[]; + minInteractivity: number; + costProvider: CostProvider; + costType: CostType; + /** Legend selection by base hwKey. Omit to consider every group. */ + visibleHwKeys?: ReadonlySet; +} + +interface Candidate { + hwKey: string; + precision: string; + vendor: string; + cost: number; + ttft: number; + interactivity: number; + point: GPUDataPoint; + runIndex?: number; +} + +function readable(point: GPUDataPoint): point is GPUDataPoint & { ttft: number } { + return typeof point.ttft === 'number' && Number.isFinite(point.ttft) && point.ttft > 0; +} + +function collectCandidates( + groups: Record, + meta: Record, + costProvider: CostProvider, + costType: CostType, + visibleHwKeys?: ReadonlySet, +): Candidate[] { + const out: Candidate[] = []; + for (const [groupKey, points] of Object.entries(groups)) { + const groupMeta = meta[groupKey]; + const hwKey = groupMeta?.hwKey ?? groupKey; + if (visibleHwKeys && !visibleHwKeys.has(hwKey)) continue; + const runIndex = groupMeta && 'runIndex' in groupMeta ? groupMeta.runIndex : undefined; + const vendor = hardwareVendor(hwKey); + for (const point of points) { + if (!readable(point)) continue; + const cost = getCostField(point, costProvider, costType); + if (!Number.isFinite(cost) || cost <= 0) continue; + if (!Number.isFinite(point.interactivity)) continue; + out.push({ + hwKey, + precision: point.precision, + vendor, + cost, + ttft: point.ttft, + interactivity: point.interactivity, + point, + ...(runIndex === undefined ? {} : { runIndex }), + }); + } + } + return out; +} + +/** + * Lower cost wins. Ties break toward the shorter first-token wait, then the + * faster stream, so a duplicate row cannot pick the worse-behaved twin. + */ +function betterCandidate(a: Candidate, b: Candidate): boolean { + if (a.cost !== b.cost) return a.cost < b.cost; + if (a.ttft !== b.ttft) return a.ttft < b.ttft; + return a.interactivity > b.interactivity; +} + +function pickWinner( + candidates: readonly Candidate[], + cap: number, + minInteractivity: number, + seriesKey: string, +): FirstTokenWinner | null { + let best: Candidate | null = null; + for (const candidate of candidates) { + if (candidate.ttft > cap || candidate.interactivity < minInteractivity) continue; + if (best === null || betterCandidate(candidate, best)) best = candidate; + } + return best ? { seriesKey, ...best } : null; +} + +/** + * For every cap in the ladder, the cheapest measured configuration per vendor + * (and per loaded unofficial run) that cleared the interactivity floor with a + * first token under the cap. + */ +export function selectFirstTokenWinners(input: FirstTokenSelectionInput): FirstTokenResult { + const { caps, minInteractivity, costProvider, costType, visibleHwKeys } = input; + + const official = collectCandidates( + input.official, + input.officialMeta, + costProvider, + costType, + visibleHwKeys, + ); + const overlay = collectCandidates( + input.overlay ?? {}, + input.overlayMeta ?? {}, + costProvider, + costType, + visibleHwKeys, + ); + + // A vendor appears as soon as it has readable rows, even when none of them + // clear a cap: an empty column says "nothing measured under this limit", + // which is the finding, and hiding it would read as the vendor not existing. + const vendors = [...new Set(official.map((c) => c.vendor))].toSorted(compareVendors); + const byVendor = new Map(); + for (const candidate of official) { + const list = byVendor.get(candidate.vendor) ?? []; + list.push(candidate); + byVendor.set(candidate.vendor, list); + } + + const runIndexes = [ + ...new Set(overlay.map((c) => c.runIndex).filter((idx): idx is number => idx !== undefined)), + ].toSorted((a, b) => a - b); + const byRun = new Map(); + for (const candidate of overlay) { + if (candidate.runIndex === undefined) continue; + const list = byRun.get(candidate.runIndex) ?? []; + list.push(candidate); + byRun.set(candidate.runIndex, list); + } + + const series: FirstTokenSeries[] = [ + ...vendors.map((vendor) => ({ key: `vendor:${vendor}`, label: vendor, vendor })), + ...runIndexes.map((runIndex) => ({ + key: `run:${runIndex}`, + label: `✕ ${input.overlayLabels?.[runIndex] ?? `run ${runIndex + 1}`}`, + runIndex, + })), + ]; + + const cells: FirstTokenCell[] = []; + const summaries: FirstTokenCapSummary[] = []; + for (const cap of caps) { + const officialWinners: FirstTokenWinner[] = []; + for (const entry of series) { + const candidates = + entry.vendor === undefined ? byRun.get(entry.runIndex!) : byVendor.get(entry.vendor); + const winner = pickWinner(candidates ?? [], cap, minInteractivity, entry.key); + cells.push({ cap, series: entry, winner }); + if (winner && entry.vendor !== undefined) officialWinners.push(winner); + } + officialWinners.sort((a, b) => a.cost - b.cost); + const best = officialWinners[0] ?? null; + const runnerUp = officialWinners[1] ?? null; + summaries.push({ + cap, + best, + runnerUp, + pctLower: + best && runnerUp && runnerUp.cost > 0 ? (runnerUp.cost - best.cost) / runnerUp.cost : null, + }); + } + + return { + series, + cells, + summaries, + qualifyingRows: official.filter((c) => c.interactivity >= minInteractivity).length, + measuredRows: official.length, + overlayMeasuredRows: overlay.length, + overlayQualifyingRows: overlay.filter((c) => c.interactivity >= minInteractivity).length, + }; +} diff --git a/packages/app/src/components/calculator/types.ts b/packages/app/src/components/calculator/types.ts index 1c2da8c95..9fec2ae52 100644 --- a/packages/app/src/components/calculator/types.ts +++ b/packages/app/src/components/calculator/types.ts @@ -13,6 +13,13 @@ export interface GPUDataPoint { sourceRow?: BenchmarkRow; hwKey: string; interactivity: number; // tokens/sec/user (median_intvty = x in interactivity chart) + /** + * Time to first token in seconds, read at the same percentile as + * `interactivity` (agentic rows) or the median (fixed sequences), mirroring + * the inference chart's TTFT axis. Absent when the row reported none, so a + * missing measurement can never pass a first-token cap as a zero. + */ + ttft?: number; /** * End-to-end latency at the selected percentile. Agentic calculator groups * use this to keep only the same anti-benchmark-hacking Pareto winners as the diff --git a/packages/app/src/components/calculator/useThroughputData.test.ts b/packages/app/src/components/calculator/useThroughputData.test.ts index 242ff4d0b..6ca4ffafd 100644 --- a/packages/app/src/components/calculator/useThroughputData.test.ts +++ b/packages/app/src/components/calculator/useThroughputData.test.ts @@ -1148,6 +1148,58 @@ describe('buildGpuGroups', () => { expect(first.tpPerMw).toBeGreaterThan(0); }); + describe('time to first token', () => { + const only = (row: BenchmarkRow, sequence = Sequence.OneK_OneK, percentile?: Percentile) => { + const { grouped } = buildGpuGroups([row], { + sequence, + precisions: ['fp4'], + percentile, + classify: singlePrecisionClassify, + }); + return Object.values(grouped)[0][0]; + }; + + it('reads the median for fixed-sequence rows, matching the inference TTFT axis', () => { + const point = only( + makeRow({ + metrics: { median_intvty: 50, tput_per_gpu: 900, median_ttft: 0.8, p90_ttft: 2.4 }, + }), + ); + expect(point.ttft).toBe(0.8); + }); + + it('reads the selected percentile for agentic rows', () => { + const agentic = (percentile: Percentile) => + only( + makeRow({ + benchmark_type: 'agentic_traces', + isl: null, + osl: null, + metrics: { + p75_itl: 1 / 200, + p90_itl: 1 / 160, + p75_e2el: 20, + p90_e2el: 30, + tput_per_gpu: 900, + median_ttft: 0.5, + p75_ttft: 1.2, + p90_ttft: 3.1, + }, + }), + Sequence.AgenticTraces, + percentile, + ); + expect(agentic(Percentile.P90).ttft).toBe(3.1); + expect(agentic(Percentile.P75).ttft).toBe(1.2); + }); + + it('is absent, not zero, when the row reported no first-token time', () => { + // A zero would clear every first-token cap; absence is filtered out. + expect(only(makeRow()).ttft).toBeUndefined(); + expect('ttft' in only(makeRow())).toBe(false); + }); + }); + describe('cached-input fraction', () => { const withMetrics = (extra: Record) => makeRow({ diff --git a/packages/app/src/components/calculator/useThroughputData.ts b/packages/app/src/components/calculator/useThroughputData.ts index af71981f0..ea3a144db 100644 --- a/packages/app/src/components/calculator/useThroughputData.ts +++ b/packages/app/src/components/calculator/useThroughputData.ts @@ -67,12 +67,31 @@ export interface OverlayGroupMeta extends GroupMeta { function getAgenticMetric( entry: AggDataEntry, percentile: Percentile, - suffix: 'intvty' | 'e2el', + suffix: 'intvty' | 'e2el' | 'ttft', ): number { const value = entry[`${percentile}_${suffix}` as keyof AggDataEntry]; return typeof value === 'number' ? value : 0; } +/** + * Time to first token for a row, at the percentile `interactivity` uses for + * agentic rows and the median for fixed sequences — the same split the + * inference chart's TTFT axis makes. Undefined rather than zero when the row + * has no usable measurement: the first-token page filters on this value, and + * a zero would clear every cap. + */ +function firstTokenSeconds( + entry: AggDataEntry, + sequence: Sequence, + percentile: Percentile, +): number | undefined { + const value = + sequence === Sequence.AgenticTraces + ? getAgenticMetric(entry, percentile, 'ttft') + : entry.median_ttft; + return typeof value === 'number' && Number.isFinite(value) && value > 0 ? value : undefined; +} + /** * Fraction of a row's input tokens served from cache, or null when the row * carries no cache metric at all — which is every fixed-sequence row. @@ -212,6 +231,8 @@ export function buildGpuGroups( if (!grouped[groupKey]) grouped[groupKey] = []; groupMeta[groupKey] = meta; + const ttft = firstTokenSeconds(entry, sequence, percentile); + grouped[groupKey].push({ sourceRow: row, hwKey, @@ -219,6 +240,7 @@ export function buildGpuGroups( sequence === Sequence.AgenticTraces ? getAgenticMetric(entry, percentile, 'intvty') : entry.median_intvty, + ...(ttft === undefined ? {} : { ttft }), ...(sequence === Sequence.AgenticTraces ? { e2eLatency: getAgenticMetric(entry, percentile, 'e2el'), diff --git a/packages/app/src/components/compare/agentx-compare-hero.tsx b/packages/app/src/components/compare/agentx-compare-hero.tsx index 535488652..31193ade7 100644 --- a/packages/app/src/components/compare/agentx-compare-hero.tsx +++ b/packages/app/src/components/compare/agentx-compare-hero.tsx @@ -436,6 +436,13 @@ export function AgentXCompareHero({ const t = STRINGS[locale]; const prefix = locale === 'zh' ? '/zh' : ''; const Heading = headingLevel; + // Landing curation must not change compare coverage or dashboard defaults. + const ledgerModels = + surface === 'landing' + ? FEATURED_AGENTX_MODELS.filter( + (model) => model.slug !== 'deepseek-v4' && model.slug !== 'qwen-3-8-flash-next', + ) + : FEATURED_AGENTX_MODELS; return (
@@ -534,7 +541,7 @@ export function AgentXCompareHero({ {/* The visible ledger header is dropped; `ledgerTitle` stays as the nav's accessible name so screen readers still get the label. */}
diff --git a/packages/app/src/components/inference/axis-metric-explanations.ts b/packages/app/src/components/inference/axis-metric-explanations.ts index 220703207..5ce643e70 100644 --- a/packages/app/src/components/inference/axis-metric-explanations.ts +++ b/packages/app/src/components/inference/axis-metric-explanations.ts @@ -431,6 +431,116 @@ export const METRIC_EXPLANATIONS: Record = { zh: '% TDP = 每芯片实测平均功耗(W)÷ 额定 TDP(W)× 100', }, }, + measuredPowerTimeline: { + description: { + en: + `The per-second accelerator power samples behind each measured average, drawn over ` + + `the whole benchmark job (server start, warmup, and the validated measurement window, ` + + `which is emphasized). One trace per config, mean of its GPUs by default; the rated TDP ` + + `is a dashed reference per hardware. Configs whose telemetry artifact is missing are ` + + `listed under the chart rather than estimated.${MEASURED_TIER_NOTE_EN}`, + zh: + `每个实测平均值背后的逐秒加速器功耗采样,覆盖整个基准测试任务(服务启动、warmup ` + + `以及被突出显示的有效测量窗口)。每个配置一条曲线,默认取其 GPU 的平均值;` + + `每种硬件的额定 TDP 以虚线作为参考。缺少遥测产物的配置会列在图表下方,而不会用估算值代替。${ + MEASURED_TIER_NOTE_ZH + }`, + }, + formula: { + en: 'W(t) = mean over GPUs of the sampled power draw in each one-second bucket', + zh: 'W(t) = 每个一秒时间桶内各 GPU 功耗采样值的平均', + }, + }, + gpuProvisionedWatts: { + description: { + en: + 'Rated accelerator TDP from the hardware registry, shown as a flat per-chip value so ' + + 'measured power can be read against the GPU-only provisioning boundary. It does not ' + + 'depend on the run.', + zh: + '取硬件注册表中的加速器额定 TDP,以每芯片恒定值显示,用于对照 GPU 侧的额定供电边界与实测功耗。' + + '该值与具体运行无关。', + }, + formula: { + en: 'W/GPU = rated TDP (W)', + zh: 'W/GPU = 额定 TDP(W)', + }, + }, + gpuProvisionedJPerOutputToken: { + description: { + en: + 'Energy per output token if every allocated accelerator drew exactly its rated TDP for ' + + 'the whole run. Disaggregated deployments count prefill and decode GPUs together, so ' + + 'this is the GPU-only provisioning boundary the measured J/token can be compared against.', + zh: + '假设所有已分配加速器在整个运行中恒以额定 TDP 耗电时的每输出 token 能耗。' + + '分离式部署将 prefill 与 decode GPU 一并计入,因此它是可与实测 J/token 对照的 GPU 侧额定边界。', + }, + formula: { + en: 'J/tok = rated TDP (W) × allocated GPUs ÷ total output tokens per second', + zh: 'J/tok = 额定 TDP(W)× 已分配 GPU 数 ÷ 总输出 token 吞吐(tok/s)', + }, + }, + utilityProvisionedWatts: { + description: { + en: + 'All-in provisioned power per chip from the hardware registry: the utility-side capacity ' + + 'a data center reserves for one accelerator including host, networking, cooling and ' + + 'power-conversion overheads. It is a flat value independent of the run.', + zh: + '取硬件注册表中的每芯片全电源配置功耗:数据中心为单张加速器预留的电源侧容量,' + + '包含主机、网络、散热与电源转换开销。该值为恒定值,与运行无关。', + }, + formula: { + en: 'W/GPU = all-in provisioned power per GPU (kW) × 1000', + zh: 'W/GPU = 每 GPU 全电源配置功耗(kW)× 1000', + }, + }, + utilityProvisionedJPerOutputToken: { + description: { + en: + 'Energy per output token at the all-in provisioned power boundary, normalized by every ' + + 'allocated accelerator. It differs from the public All-in Provisioned J per Output Token ' + + 'metric only for disaggregated runs, where that metric normalizes by decode GPUs alone.', + zh: + '在全电源配置边界下的每输出 token 能耗,按全部已分配加速器归一。' + + '仅在分离式运行中与公开的 All-in Provisioned J per Output Token 指标不同,后者只按 decode GPU 归一。', + }, + formula: { + en: 'J/tok = all-in provisioned power per GPU (W) × allocated GPUs ÷ total output tokens per second', + zh: 'J/tok = 每 GPU 全电源配置功耗(W)× 已分配 GPU 数 ÷ 总输出 token 吞吐(tok/s)', + }, + }, + utilityModeledWatts: { + description: { + en: + 'Modeled facility power per allocated accelerator: measured GPU power is scaled to chassis ' + + 'AC by the system power model and then multiplied once by PUE. Only hardware with a known ' + + 'eight-GPU chassis profile on 8k1k runs is supported; NVL72 systems show no value.', + zh: + '每已分配加速器的数据中心建模功耗:先由系统功耗模型将 GPU 实测功耗换算为机箱交流功耗,再乘以一次 PUE。' + + '仅支持在 8k1k 运行中具有已知八卡机箱模型的硬件;NVL72 系统不显示数值。', + }, + formula: { + en: 'W/GPU = modeled chassis AC power (W) × PUE ÷ allocated GPUs', + zh: 'W/GPU = 机箱交流建模功耗(W)× PUE ÷ 已分配 GPU 数', + }, + }, + utilityModeledJPerOutputToken: { + description: { + en: + 'Measured energy per output token scaled to the modeled facility boundary, so its ratio to ' + + 'measured GPU energy equals the ratio of modeled facility power to measured GPU power. ' + + 'Missing where the system power model or validated measured power is unavailable.', + zh: + '将实测每输出 token 能耗按建模的数据中心边界缩放,其与 GPU 实测能耗之比等于数据中心建模功耗与 GPU 实测功耗之比。' + + '系统功耗模型或通过验证的实测功耗缺失时不显示。', + }, + formula: { + en: 'J/tok = measured J per output token × modeled facility W per GPU ÷ measured W per GPU', + zh: 'J/tok = 实测每输出 token 能耗 × 每 GPU 数据中心建模功耗(W)÷ 每 GPU 实测功耗(W)', + }, + }, }; /** diff --git a/packages/app/src/components/inference/hooks/useChartData.ts b/packages/app/src/components/inference/hooks/useChartData.ts index fecdfcd0f..7edd3955d 100644 --- a/packages/app/src/components/inference/hooks/useChartData.ts +++ b/packages/app/src/components/inference/hooks/useChartData.ts @@ -11,6 +11,7 @@ import type { ChartDefinition, HardwareConfig, InferenceData, + PowerCompare, RenderableGraph, TokenRevenuePricing, TokenRevenuePriceSource, @@ -21,6 +22,7 @@ import { usesTokenSalePricing, } from '@/components/inference/token-revenue'; import { partitionChartDataByLimits } from '@/components/inference/utils'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; import { parseComparisonEntry, resolveComparisonEntries, @@ -270,6 +272,8 @@ export function useChartData( tcoBasis: TcoBasis = DEFAULT_TCO_BASIS, /** Opt-in from the inference page only; explicit date/run/history views opt out. */ allowDefaultRunPreference = false, + /** Sibling boundary / role series appended to a gated power metric (`i_pcompare`). */ + powerCompare: PowerCompare = 'none', ) { // When the selected date is the latest available, use '' (empty string) to match // the initial no-date query key, reusing the eagerly-fetched benchmarks from the @@ -692,8 +696,14 @@ export function useChartData( ); const hasMetric = metricData.length > 0; const isTtftX = typeof xAxisField === 'string' && xAxisField.endsWith('_ttft'); + // Comparison clones are appended after the remap so they share the + // base point's x and differ only in y and `powerVariant`. const mappedData = hasMetric - ? metricData.map((d) => remapInferencePoint(d, metricKey, xAxisField)) + ? expandPowerCompareSeries( + metricData.map((d) => remapInferencePoint(d, metricKey, xAxisField)), + selectedYAxisMetric, + powerCompare, + ) : []; const isAgentic = selectedSequence === Sequence.AgenticTraces; @@ -730,6 +740,7 @@ export function useChartData( compareGpuPair, selectedPercentile, quickFilters, + powerCompare, ]); // Points that pass every scope filter but NOT the y-metric coverage filter. diff --git a/packages/app/src/components/inference/measured-metric-config.test.ts b/packages/app/src/components/inference/measured-metric-config.test.ts index 47c356f22..f7b4de74c 100644 --- a/packages/app/src/components/inference/measured-metric-config.test.ts +++ b/packages/app/src/components/inference/measured-metric-config.test.ts @@ -1,6 +1,11 @@ import { describe, expect, it } from 'vitest'; -import { MEASURED_ENERGY_METRIC_CONFIG_KEYS, METRIC_CONFIG_KEYS } from './metric-registry'; +import { + MEASURED_ENERGY_METRIC_CONFIG_KEYS, + METRIC_CONFIG_KEYS, + POWER_BASIS_METRIC_CONFIG_KEYS, +} from './metric-registry'; +import { POWER_BASES } from '@/lib/power-basis'; import { changeMeasuredMetricConfig, getMeasuredMetricConfig, @@ -18,10 +23,23 @@ describe('measured metric configuration', () => { }, ); + it.each(POWER_BASIS_METRIC_CONFIG_KEYS)('round-trips the power-boundary metric %s', (key) => { + const config = getMeasuredMetricConfig(key); + expect(config).toBeDefined(); + expect(config?.basis).not.toBe('gpu-measured'); + expect(changeMeasuredMetricConfig(key, {})).toBe(key); + expect(changeMeasuredMetricConfig('y_tpPerGpu', config!)).toBe(key); + }); + it('does not group unrelated metrics or unknown persisted values', () => { const grouped = METRIC_CONFIG_KEYS.filter((key) => getMeasuredMetricConfig(key)); - expect(grouped).toHaveLength(13); - expect(new Set(grouped)).toEqual(new Set(MEASURED_ENERGY_METRIC_CONFIG_KEYS)); + expect(grouped).toHaveLength(20); + expect(new Set(grouped)).toEqual( + new Set([...MEASURED_ENERGY_METRIC_CONFIG_KEYS, ...POWER_BASIS_METRIC_CONFIG_KEYS]), + ); + for (const key of MEASURED_ENERGY_METRIC_CONFIG_KEYS) { + expect(getMeasuredMetricConfig(key)?.basis, key).toBe('gpu-measured'); + } expect(getMeasuredMetricConfig('y_modeledChassisPowerPerGpu')).toBeUndefined(); expect(getMeasuredMetricConfig('y_removedMetric')).toBeUndefined(); expect(getMeasuredMetricConfig('')).toBeUndefined(); @@ -42,6 +60,7 @@ describe('measured metric configuration', () => { it('keeps fleet percentiles, role averages and TDP normalization distinct', () => { expect(getMeasuredMetricConfig('y_measuredP90Power')).toEqual({ family: 'power', + basis: 'gpu-measured', scope: 'all', statistic: 'p90', display: 'watts', @@ -69,6 +88,36 @@ describe('measured metric configuration', () => { ).toBe('y_measuredDecodeAvgPower'); }); + it('offers the telemetry timeline only for the whole-deployment average', () => { + expect(getMeasuredMetricConfig('y_measuredPowerTimeline')).toEqual({ + family: 'power', + basis: 'gpu-measured', + scope: 'all', + statistic: 'average', + display: 'timeline', + }); + expect(changeMeasuredMetricConfig('y_measuredAvgPower', { display: 'timeline' })).toBe( + 'y_measuredPowerTimeline', + ); + expect(changeMeasuredMetricConfig('y_measuredPowerPercentTdp', { display: 'timeline' })).toBe( + 'y_measuredPowerTimeline', + ); + // Percentiles and role scopes have no per-second trace; they fall back to watts. + expect(changeMeasuredMetricConfig('y_measuredPowerTimeline', { statistic: 'p90' })).toBe( + 'y_measuredP90Power', + ); + expect(changeMeasuredMetricConfig('y_measuredPowerTimeline', { scope: 'prefill' })).toBe( + 'y_measuredPrefillAvgPower', + ); + expect(changeMeasuredMetricConfig('y_measuredP75Power', { display: 'timeline' })).toBe( + 'y_measuredP75Power', + ); + // Leaving the timeline for energy and coming back lands on the family default. + expect(changeMeasuredMetricConfig('y_measuredPowerTimeline', { family: 'energy' })).toBe( + 'y_measuredJPerOutputToken', + ); + }); + it('changes the energy denominator without silently attributing whole-run energy to a role', () => { expect(changeMeasuredMetricConfig('y_measuredJPerOutputToken', { denominator: 'input' })).toBe( 'y_measuredJPerInputToken', @@ -78,6 +127,7 @@ describe('measured metric configuration', () => { ); expect(getMeasuredMetricConfig('y_measuredPrefillJPerInputToken')).toEqual({ family: 'energy', + basis: 'gpu-measured', scope: 'prefill', denominator: 'input', unit: 'joules', @@ -110,4 +160,93 @@ describe('measured metric configuration', () => { changeMeasuredMetricConfig('y_measuredPrefillJPerInputToken', { unit: 'wattHours' }), ).toBe('y_measuredPrefillJPerInputToken'); }); + + it('keeps the existing share-link defaults on the GPU-measured boundary', () => { + expect(getMeasuredMetricConfig(MEASURED_METRIC_DEFAULTS.power)?.basis).toBe('gpu-measured'); + expect(getMeasuredMetricConfig(MEASURED_METRIC_DEFAULTS.energy)?.basis).toBe('gpu-measured'); + expect(POWER_BASES[0]).toBe('gpu-measured'); + }); + + it('snaps a boundary change to that boundary’s canonical combination', () => { + // The P90 statistic has no provisioned counterpart: choosing the boundary + // moves to its whole-deployment average watts. + expect(changeMeasuredMetricConfig('y_measuredP90Power', { basis: 'gpu-provisioned' })).toBe( + 'y_gpuProvisionedWatts', + ); + expect( + changeMeasuredMetricConfig('y_measuredPowerPercentTdp', { basis: 'utility-provisioned' }), + ).toBe('y_utilityProvisionedWatts'); + expect( + changeMeasuredMetricConfig('y_measuredDecodeAvgPower', { basis: 'utility-modeled' }), + ).toBe('y_utilityModeledWatts'); + // Role- and query-scoped energy snap to joules per output token. + expect( + changeMeasuredMetricConfig('y_measuredPrefillJPerInputToken', { basis: 'utility-modeled' }), + ).toBe('y_utilityModeledJPerOutputToken'); + expect( + changeMeasuredMetricConfig('y_measuredWhPerSuccessfulQuery', { basis: 'gpu-provisioned' }), + ).toBe('y_gpuProvisionedJPerOutputToken'); + // Boundary-to-boundary moves stay within the family. + expect( + changeMeasuredMetricConfig('y_gpuProvisionedWatts', { basis: 'utility-provisioned' }), + ).toBe('y_utilityProvisionedWatts'); + expect( + changeMeasuredMetricConfig('y_utilityModeledJPerOutputToken', { basis: 'gpu-provisioned' }), + ).toBe('y_gpuProvisionedJPerOutputToken'); + // A boundary requested together with other dimensions wins over them. + expect( + changeMeasuredMetricConfig('y_measuredAvgPower', { + basis: 'utility-modeled', + statistic: 'p90', + }), + ).toBe('y_utilityModeledWatts'); + }); + + it('returns to GPU-measured telemetry from a boundary', () => { + expect(changeMeasuredMetricConfig('y_gpuProvisionedWatts', { basis: 'gpu-measured' })).toBe( + 'y_measuredAvgPower', + ); + expect( + changeMeasuredMetricConfig('y_utilityProvisionedJPerOutputToken', { basis: 'gpu-measured' }), + ).toBe('y_measuredJPerOutputToken'); + // Every other dimension exists only for telemetry, so changing one from a + // boundary lands on the nearest GPU-measured key rather than a dead end. + expect(changeMeasuredMetricConfig('y_utilityModeledWatts', { statistic: 'p75' })).toBe( + 'y_measuredP75Power', + ); + expect(changeMeasuredMetricConfig('y_gpuProvisionedWatts', { scope: 'decode' })).toBe( + 'y_measuredDecodeAvgPower', + ); + expect(changeMeasuredMetricConfig('y_utilityProvisionedWatts', { display: 'tdp' })).toBe( + 'y_measuredPowerPercentTdp', + ); + expect( + changeMeasuredMetricConfig('y_utilityModeledJPerOutputToken', { denominator: 'query' }), + ).toBe('y_measuredJPerSuccessfulQuery'); + expect(changeMeasuredMetricConfig('y_gpuProvisionedJPerOutputToken', { scope: 'decode' })).toBe( + 'y_measuredDecodeJPerOutputToken', + ); + // Re-selecting the canonical value is still a dimension change and leaves the boundary. + expect(changeMeasuredMetricConfig('y_utilityModeledWatts', { statistic: 'average' })).toBe( + 'y_measuredAvgPower', + ); + expect(changeMeasuredMetricConfig('y_gpuProvisionedJPerOutputToken', { unit: 'joules' })).toBe( + 'y_measuredJPerOutputToken', + ); + }); + + it('keeps the boundary across a family switch', () => { + expect(changeMeasuredMetricConfig('y_utilityProvisionedWatts', { family: 'energy' })).toBe( + 'y_utilityProvisionedJPerOutputToken', + ); + expect(changeMeasuredMetricConfig('y_gpuProvisionedJPerOutputToken', { family: 'power' })).toBe( + 'y_gpuProvisionedWatts', + ); + expect( + changeMeasuredMetricConfig('y_tokensPerDollarH', { + family: 'energy', + basis: 'utility-modeled', + }), + ).toBe('y_utilityModeledJPerOutputToken'); + }); }); diff --git a/packages/app/src/components/inference/measured-metric-config.ts b/packages/app/src/components/inference/measured-metric-config.ts index 555fbef2e..ff28546be 100644 --- a/packages/app/src/components/inference/measured-metric-config.ts +++ b/packages/app/src/components/inference/measured-metric-config.ts @@ -1,17 +1,27 @@ +import type { PowerBasis } from '@/lib/power-basis'; import type { MetricConfigKey } from './metric-registry'; export type MeasuredMetricFamily = 'power' | 'energy'; type MeasuredScope = 'all' | 'prefill' | 'decode'; +/** + * How whole-deployment average power is shown: per-chip watts, percent of + * TDP, or the per-second telemetry trace behind the average (`timeline`, which + * ChartDisplay renders with `PowerTimeline` instead of the scatter chart). + */ +export type MeasuredPowerDisplay = 'watts' | 'tdp' | 'timeline'; export type MeasuredMetricConfig = | { family: 'power'; + /** Power boundary the key plots; only `gpu-measured` publishes the other dimensions. */ + basis: PowerBasis; scope: MeasuredScope; statistic: 'average' | 'p75' | 'p90'; - display: 'watts' | 'tdp'; + display: MeasuredPowerDisplay; } | { family: 'energy'; + basis: PowerBasis; scope: MeasuredScope; denominator: 'input' | 'output' | 'total' | 'query'; unit: 'joules' | 'wattHours'; @@ -19,9 +29,10 @@ export type MeasuredMetricConfig = export type MeasuredMetricConfigChange = Partial<{ family: MeasuredMetricFamily; + basis: PowerBasis; scope: MeasuredScope; statistic: 'average' | 'p75' | 'p90'; - display: 'watts' | 'tdp'; + display: MeasuredPowerDisplay; denominator: 'input' | 'output' | 'total' | 'query'; unit: 'joules' | 'wattHours'; }>; @@ -31,51 +42,78 @@ export const MEASURED_METRIC_DEFAULTS = { energy: 'y_measuredJPerOutputToken', } as const satisfies Record; +const measured = { basis: 'gpu-measured' } as const; + // Presentation settings resolve to existing metrics; they do not own chart state. const MEASURED_METRIC_CONFIGS: readonly (readonly [MetricConfigKey, MeasuredMetricConfig])[] = [ - ['y_measuredAvgPower', { family: 'power', scope: 'all', statistic: 'average', display: 'watts' }], - ['y_measuredP75Power', { family: 'power', scope: 'all', statistic: 'p75', display: 'watts' }], - ['y_measuredP90Power', { family: 'power', scope: 'all', statistic: 'p90', display: 'watts' }], + [ + 'y_measuredAvgPower', + { family: 'power', ...measured, scope: 'all', statistic: 'average', display: 'watts' }, + ], + [ + 'y_measuredP75Power', + { family: 'power', ...measured, scope: 'all', statistic: 'p75', display: 'watts' }, + ], + [ + 'y_measuredP90Power', + { family: 'power', ...measured, scope: 'all', statistic: 'p90', display: 'watts' }, + ], [ 'y_measuredPrefillAvgPower', - { family: 'power', scope: 'prefill', statistic: 'average', display: 'watts' }, + { family: 'power', ...measured, scope: 'prefill', statistic: 'average', display: 'watts' }, ], [ 'y_measuredDecodeAvgPower', - { family: 'power', scope: 'decode', statistic: 'average', display: 'watts' }, + { family: 'power', ...measured, scope: 'decode', statistic: 'average', display: 'watts' }, ], [ 'y_measuredPowerPercentTdp', - { family: 'power', scope: 'all', statistic: 'average', display: 'tdp' }, + { family: 'power', ...measured, scope: 'all', statistic: 'average', display: 'tdp' }, + ], + [ + 'y_measuredPowerTimeline', + { family: 'power', ...measured, scope: 'all', statistic: 'average', display: 'timeline' }, ], [ 'y_measuredJPerInputToken', - { family: 'energy', scope: 'all', denominator: 'input', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'input', unit: 'joules' }, ], [ 'y_measuredJPerOutputToken', - { family: 'energy', scope: 'all', denominator: 'output', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'output', unit: 'joules' }, ], [ 'y_measuredJPerTotalToken', - { family: 'energy', scope: 'all', denominator: 'total', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'total', unit: 'joules' }, ], [ 'y_measuredPrefillJPerInputToken', - { family: 'energy', scope: 'prefill', denominator: 'input', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'prefill', denominator: 'input', unit: 'joules' }, ], [ 'y_measuredDecodeJPerOutputToken', - { family: 'energy', scope: 'decode', denominator: 'output', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'decode', denominator: 'output', unit: 'joules' }, ], [ 'y_measuredJPerSuccessfulQuery', - { family: 'energy', scope: 'all', denominator: 'query', unit: 'joules' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'query', unit: 'joules' }, ], [ 'y_measuredWhPerSuccessfulQuery', - { family: 'energy', scope: 'all', denominator: 'query', unit: 'wattHours' }, + { family: 'energy', ...measured, scope: 'all', denominator: 'query', unit: 'wattHours' }, ], + // Derived boundaries publish one canonical combination per family: whole + // deployment, average watts, joules per output token (lib/power-basis.ts). + ...( + [ + ['gpu-provisioned', 'y_gpuProvisionedWatts', 'y_gpuProvisionedJPerOutputToken'], + ['utility-provisioned', 'y_utilityProvisionedWatts', 'y_utilityProvisionedJPerOutputToken'], + ['utility-modeled', 'y_utilityModeledWatts', 'y_utilityModeledJPerOutputToken'], + ] as const satisfies readonly (readonly [PowerBasis, MetricConfigKey, MetricConfigKey])[] + ).flatMap(([basis, watts, energy]): (readonly [MetricConfigKey, MeasuredMetricConfig])[] => [ + [watts, { family: 'power', basis, scope: 'all', statistic: 'average', display: 'watts' }], + [energy, { family: 'energy', basis, scope: 'all', denominator: 'output', unit: 'joules' }], + ]), ]; export function getMeasuredMetricConfig(metric: string): MeasuredMetricConfig | undefined { @@ -83,6 +121,8 @@ export function getMeasuredMetricConfig(metric: string): MeasuredMetricConfig | return config ? { ...config } : undefined; } +const OTHER_DIMENSIONS = ['scope', 'statistic', 'display', 'denominator', 'unit'] as const; + export function changeMeasuredMetricConfig( metric: string, change: MeasuredMetricConfigChange, @@ -93,6 +133,21 @@ export function changeMeasuredMetricConfig( current?.family === family ? current : getMeasuredMetricConfig(MEASURED_METRIC_DEFAULTS[family])!; + // Choosing a derived boundary snaps the other dimensions to its canonical + // combination. Changing any of those dimensions while on a derived boundary + // returns to GPU-measured telemetry, the only basis that publishes variants, + // so every control change lands on a real key. A family switch keeps the + // boundary: the metric key is what carries it. + const changesOtherDimension = OTHER_DIMENSIONS.some((key) => change[key] !== undefined); + const basis = + change.basis ?? (changesOtherDimension ? 'gpu-measured' : (current ?? config).basis); + if (basis !== 'gpu-measured') { + return ( + MEASURED_METRIC_CONFIGS.find( + ([, candidate]) => candidate.family === family && candidate.basis === basis, + )?.[0] ?? MEASURED_METRIC_DEFAULTS[family] + ); + } let scope = change.scope ?? config.scope; if (config.family === 'power') { @@ -103,6 +158,7 @@ export function changeMeasuredMetricConfig( MEASURED_METRIC_CONFIGS.find( ([, candidate]) => candidate.family === 'power' && + candidate.basis === 'gpu-measured' && candidate.scope === scope && candidate.statistic === statistic && candidate.display === display, @@ -123,6 +179,7 @@ export function changeMeasuredMetricConfig( MEASURED_METRIC_CONFIGS.find( ([, candidate]) => candidate.family === 'energy' && + candidate.basis === 'gpu-measured' && candidate.scope === scope && candidate.denominator === denominator && candidate.unit === unit, diff --git a/packages/app/src/components/inference/measured-power-direction.test.ts b/packages/app/src/components/inference/measured-power-direction.test.ts index 44d2e4ff0..790df173f 100644 --- a/packages/app/src/components/inference/measured-power-direction.test.ts +++ b/packages/app/src/components/inference/measured-power-direction.test.ts @@ -3,6 +3,11 @@ import { describe, it, expect } from 'vitest'; import chartDefinitions from '@/components/inference/metric-registry'; import type { ChartDefinition, InferenceData } from '@/components/inference/types'; import { sortRowsByYMetric } from '@/components/inference/ui/inference-table-sort'; +import { + isMeasuredPowerCurveMetric, + isPowerCurveMetric, + upperPowerEnvelope, +} from '@/components/inference/utils/powerCurves'; import { isFrontierEligible, paretoFrontForDirection, @@ -29,6 +34,24 @@ const MEASURED_POWER_METRICS = [ 'y_measuredPrefillAvgPower', 'y_measuredDecodeAvgPower', 'y_measuredPowerPercentTdp', + // Derived power boundaries share the watts semantics: the declared corner + // drives the ascending table sort, while the chart itself draws the upper + // power envelope (asserted separately below), exactly like measured watts. + 'y_gpuProvisionedWatts', + 'y_utilityProvisionedWatts', + 'y_utilityModeledWatts', +] as const; + +const BASIS_WATT_METRICS = [ + 'y_gpuProvisionedWatts', + 'y_utilityProvisionedWatts', + 'y_utilityModeledWatts', +] as const; + +const BASIS_ENERGY_METRICS = [ + 'y_gpuProvisionedJPerOutputToken', + 'y_utilityProvisionedJPerOutputToken', + 'y_utilityModeledJPerOutputToken', ] as const; const QUERY_ENERGY_METRICS = [ @@ -165,6 +188,46 @@ describe('measured-power Pareto direction', () => { } }); + it.each(BASIS_ENERGY_METRICS)('%s is bilingual and lower-is-better', (metric) => { + expect(declaredDirection(interactivityDef, metric)).toBe('lower_right'); + expect(declaredDirection(e2eDef, metric)).toBe('lower_left'); + for (const chartDef of [interactivityDef, e2eDef]) { + expect(chartDef[metric]).toMatch(/\.y$/u); + expect(chartDef[`${metric}_label`]).toBeTruthy(); + expect(chartDef[`${metric}_labelZh`]).toBeTruthy(); + } + // Energy uses the Pareto corner, not the power envelope: the dominated + // sweep points drop out just as they do for measured joules. + expect(frontierConcs(interactivityDef, metric)).toEqual([1, 32]); + expect(frontierConcs(e2eDef, metric)).toEqual([256, 32]); + }); + + describe.each(BASIS_WATT_METRICS)('%s', (metric) => { + it('draws the upper power envelope, not the Pareto corner, like measured watts', () => { + // ScatterGraph/GPUGraph pick the curve by `isPowerCurveMetric`; a watt + // boundary that misses that set would hide every dominated sweep point + // under Optimal Only and, for a flat TDP series, collapse to one marker. + expect(isPowerCurveMetric(metric)).toBe(true); + expect(isPowerCurveMetric('y_measuredAvgPower')).toBe(true); + // Envelope-locked like measured watts: with Optimal Only on, a Pareto + // corner would keep one marker and draw no curve for a flat TDP series. + expect(isMeasuredPowerCurveMetric(metric)).toBe(true); + expect(isMeasuredPowerCurveMetric('y_modeledChassisPowerPerGpu')).toBe(false); + expect(isMeasuredPowerCurveMetric('y_measuredAvgPower')).toBe(true); + + const metricField = (interactivityDef[metric] as string).replace(/\.y$/u, ''); + const envelope = upperPowerEnvelope(sweepPoints(metricField), true).map((p) => p.conc); + // The 1400 W conc 8 peak survives on the envelope; the Pareto corner drops it. + expect(envelope).toContain(8); + expect(frontierConcs(interactivityDef, metric)).not.toContain(8); + }); + }); + + it.each(BASIS_ENERGY_METRICS)('%s stays off the power-envelope path', (metric) => { + expect(isPowerCurveMetric(metric)).toBe(false); + expect(isMeasuredPowerCurveMetric(metric)).toBe(false); + }); + it('keeps %TDP bilingual while using the same per-hardware frontier as watts', () => { // A fixed hardware TDP rescales watts without changing dominance within // that hardware series. The shared sweep tests above exercise both axes. diff --git a/packages/app/src/components/inference/metric-registry.test.ts b/packages/app/src/components/inference/metric-registry.test.ts index 28045728e..c2944fa84 100644 --- a/packages/app/src/components/inference/metric-registry.test.ts +++ b/packages/app/src/components/inference/metric-registry.test.ts @@ -10,6 +10,7 @@ import { isBenchmarkMetricKey, isMeasuredEnergyConfigKey, isModeledSystemPowerConfigKey, + isPowerBasisConfigKey, isRoleLocalMeasuredEnergyConfigKey, MEASURED_ENERGY_METRIC_CONFIG_KEYS, METRIC_CONFIG_KEYS, @@ -19,9 +20,11 @@ import { metricCostTier, metricForCostTier, metricOptionTitle, + POWER_BASIS_METRIC_CONFIG_KEYS, resolveMetricConfigKey, tokenMetricTypeForConfigKey, } from './metric-registry'; +import { POWER_BASIS_FIELDS } from '@/lib/power-basis'; import type { YAxisMetricKey } from './types'; describe('metric registry', () => { @@ -39,6 +42,10 @@ describe('metric registry', () => { expect(e2e.y_costh_roofline).toBe('lower_left'); expect(interactivity.y_measuredPowerPercentTdp_roofline).toBe('lower_right'); expect(e2e.y_measuredPowerPercentTdp_roofline).toBe('lower_left'); + for (const key of POWER_BASIS_METRIC_CONFIG_KEYS) { + expect(interactivity[`${key}_roofline`], key).toBe('lower_right'); + expect(e2e[`${key}_roofline`], key).toBe('lower_left'); + } }); it('preserves metric-specific x overrides and bilingual labels', () => { @@ -210,7 +217,52 @@ describe('metric registry', () => { ); const measuredGroup = METRIC_CONTROL_GROUPS.find((group) => group.label === 'Measured Energy'); - expect(measuredGroup?.metrics).toBe(MEASURED_ENERGY_METRIC_CONFIG_KEYS); + expect(measuredGroup?.gated).toBe(true); + expect(measuredGroup?.metrics).toEqual([ + ...MEASURED_ENERGY_METRIC_CONFIG_KEYS, + ...POWER_BASIS_METRIC_CONFIG_KEYS, + ]); + }); + + it('files the derived power boundaries under the gate without a telemetry tier', () => { + // Every derived field lib/power-basis.ts can emit is selectable from the + // gated group, and none of them is mistaken for runner telemetry. + const basisFields = Object.values(POWER_BASIS_FIELDS).flatMap((fields) => + Object.values(fields).map((field) => `y_${field}`), + ); + expect([...POWER_BASIS_METRIC_CONFIG_KEYS].toSorted()).toEqual(basisFields.toSorted()); + for (const key of POWER_BASIS_METRIC_CONFIG_KEYS) { + expect(key, key).not.toMatch(/^y_measured/u); + expect(isPowerBasisConfigKey(key), key).toBe(true); + expect(isMeasuredEnergyConfigKey(key), key).toBe(false); + expect(isModeledSystemPowerConfigKey(key), key).toBe(false); + expect(resolveMetricConfigKey(key), key).toBe(key); + const metric = METRIC_REGISTRY[key.slice(2) as keyof typeof METRIC_REGISTRY]; + expect(chartDefinitions[0][key], key).toBe(metric.field); + expect(metric.labelZh, key).toMatch(/\p{Script=Han}/u); + expect(metric.titleZh, key).toMatch(/\p{Script=Han}/u); + } + for (const key of MEASURED_ENERGY_METRIC_CONFIG_KEYS) { + expect(isPowerBasisConfigKey(key), key).toBe(false); + } + expect(isPowerBasisConfigKey('y_modeledChassisPowerPerGpu')).toBe(false); + expect(isPowerBasisConfigKey('y_jOutput')).toBe(false); + // Watts keys stay token-agnostic; energy keys are output-token metrics, so + // the output-capable point filter admits them. + expect(tokenMetricTypeForConfigKey('y_utilityProvisionedWatts')).toBe('total'); + expect(tokenMetricTypeForConfigKey('y_utilityProvisionedJPerOutputToken')).toBe('output'); + expect(tokenMetricTypeForConfigKey('y_utilityModeledJPerOutputToken')).toBe('output'); + }); + + it('keeps the all-GPU utility energy distinguishable from the per-decode-GPU jOutput', () => { + const labels = [ + METRIC_REGISTRY.jOutput.label, + METRIC_REGISTRY.utilityProvisionedJPerOutputToken.label, + METRIC_REGISTRY.jOutput.labelZh, + METRIC_REGISTRY.utilityProvisionedJPerOutputToken.labelZh, + ]; + expect(new Set(labels).size).toBe(labels.length); + expect(METRIC_REGISTRY.utilityProvisionedJPerOutputToken.title).toContain('all GPUs'); }); it('classifies measured-energy config keys', () => { diff --git a/packages/app/src/components/inference/metric-registry.ts b/packages/app/src/components/inference/metric-registry.ts index ccba918b1..9687c561d 100644 --- a/packages/app/src/components/inference/metric-registry.ts +++ b/packages/app/src/components/inference/metric-registry.ts @@ -386,6 +386,80 @@ export const METRIC_REGISTRY = { titleZh: '实测平均功耗占 TDP 百分比', polarity: 'lower', }, + // The per-second telemetry behind `measuredAvgPower`. The field aliases the + // same average so the table view, availability panel, and share links keep + // working; ChartDisplay swaps the scatter chart for `PowerTimeline`, which + // fetches each point's `gpu_metrics_*` artifact and draws the trace. The + // label leads with "Measured Average Power", like the %TDP display, so a + // "Measured Power" search still finds only the family option. + measuredPowerTimeline: { + field: 'measuredPowerTimeline.y', + label: 'Measured Average Power per Chip over Time (W)', + labelZh: '每芯片实测平均功耗时间线(W)', + title: 'Measured Average Power per Chip over Time', + titleZh: '每芯片实测平均功耗时间线', + polarity: 'lower', + }, + // Power boundaries beyond GPU-measured telemetry (`lib/power-basis.ts`). + // Each boundary publishes W per allocated GPU and J per output token; the + // Boundary select in the Measured controls resolves to these keys, so the + // metric key alone carries the boundary in share links. Keys deliberately + // lack the `measured` prefix: they are spec constants or model output, not + // telemetry, so the telemetry-only decorations must not treat them as such. + gpuProvisionedWatts: { + field: 'gpuProvisionedWatts.y', + label: 'GPU Provisioned Power per Chip (TDP, W)', + labelZh: '每芯片 GPU 额定功耗(TDP,W)', + title: 'GPU Provisioned Power per Chip (TDP)', + titleZh: '每芯片 GPU 额定功耗(TDP)', + polarity: 'lower', + }, + gpuProvisionedJPerOutputToken: { + field: 'gpuProvisionedJPerOutputToken.y', + label: 'GPU Provisioned J per Output Token (TDP, J/tok)', + labelZh: '每输出 token GPU 额定能耗(TDP,J/tok)', + title: 'GPU Provisioned Joules per Output Token (TDP)', + titleZh: '每输出 token GPU 额定焦耳能耗(TDP)', + polarity: 'lower', + }, + utilityProvisionedWatts: { + field: 'utilityProvisionedWatts.y', + label: 'Utility Provisioned Power per Chip (all-in, W)', + labelZh: '每芯片全电源配置功耗(all-in,W)', + title: 'Utility Provisioned Power per Chip (all-in)', + titleZh: '每芯片全电源配置功耗(all-in)', + polarity: 'lower', + }, + // Unlike the ungated `jOutput`, which divides by output tokens per decode + // GPU, this normalizes by every allocated GPU (prefill + decode). + utilityProvisionedJPerOutputToken: { + field: 'utilityProvisionedJPerOutputToken.y', + label: 'Utility Provisioned J per Output Token, all GPUs (all-in, J/tok)', + labelZh: '每输出 token 全电源配置能耗,按全部 GPU 归一(all-in,J/tok)', + title: 'Utility Provisioned Joules per Output Token, all GPUs (all-in)', + titleZh: '每输出 token 全电源配置焦耳能耗,按全部 GPU 归一(all-in)', + polarity: 'lower', + }, + // zh vocabulary shared with the Boundary select, its help text and the chart + // caption: B3 “全电源配置” (as the ungated jOutput/jTotal already say for + // all-in), B4 “数据中心建模” (measured GPU power carried through the chassis + // model to the utility meter). + utilityModeledWatts: { + field: 'utilityModeledWatts.y', + label: 'Utility Modeled Power per Chip (PUE, W)', + labelZh: '每芯片数据中心建模功耗(含 PUE,W)', + title: 'Utility Modeled Power per Chip (PUE)', + titleZh: '每芯片数据中心建模功耗(含 PUE)', + polarity: 'lower', + }, + utilityModeledJPerOutputToken: { + field: 'utilityModeledJPerOutputToken.y', + label: 'Utility Modeled J per Output Token (PUE, J/tok)', + labelZh: '每输出 token 数据中心建模能耗(含 PUE,J/tok)', + title: 'Utility Modeled Joules per Output Token (PUE)', + titleZh: '每输出 token 数据中心建模焦耳能耗(含 PUE)', + polarity: 'lower', + }, } as const satisfies Record; export type MetricKey = keyof typeof METRIC_REGISTRY; @@ -597,6 +671,7 @@ export const MEASURED_ENERGY_METRIC_CONFIG_KEYS = [ 'y_measuredJPerSuccessfulQuery', 'y_measuredWhPerSuccessfulQuery', 'y_measuredPowerPercentTdp', + 'y_measuredPowerTimeline', ] as const satisfies readonly MetricConfigKey[]; const MEASURED_ENERGY_METRIC_CONFIG_KEY_SET: ReadonlySet = new Set( @@ -618,6 +693,31 @@ export function isRoleLocalMeasuredEnergyConfigKey(configKey: string): boolean { return ROLE_LOCAL_MEASURED_ENERGY_METRIC_CONFIG_KEY_SET.has(configKey); } +/** + * The derived power-boundary y-axes (GPU provisioned, utility provisioned, + * utility modeled) that share the gated Measured Energy group and its + * Boundary select. They are kept out of `MEASURED_ENERGY_METRIC_CONFIG_KEYS` + * on purpose: spec constants and model output carry no telemetry tier, so the + * legacy-power ring, tier tooltip line, and footer key do not apply to them. + */ +export const POWER_BASIS_METRIC_CONFIG_KEYS = [ + 'y_gpuProvisionedWatts', + 'y_gpuProvisionedJPerOutputToken', + 'y_utilityProvisionedWatts', + 'y_utilityProvisionedJPerOutputToken', + 'y_utilityModeledWatts', + 'y_utilityModeledJPerOutputToken', +] as const satisfies readonly MetricConfigKey[]; + +const POWER_BASIS_METRIC_CONFIG_KEY_SET: ReadonlySet = new Set( + POWER_BASIS_METRIC_CONFIG_KEYS, +); + +/** Whether a y-axis config key plots a derived power boundary (B2–B4). */ +export function isPowerBasisConfigKey(configKey: string): boolean { + return POWER_BASIS_METRIC_CONFIG_KEY_SET.has(configKey); +} + export const MODELED_SYSTEM_POWER_METRIC_CONFIG_KEY = 'y_modeledChassisPowerPerGpu'; /** Whether a y-axis config key plots the modeled chassis AC power metric. */ @@ -671,10 +771,13 @@ export const METRIC_CONTROL_GROUPS: readonly MetricControlGroup[] = [ // Runner power telemetry and the chassis model built on it are still being // validated, so both groups stay behind the ↑↑↓↓ feature gate until the // measurements are stable enough to publish. + // The derived boundaries ride along so the same gate and the same + // shared-URL exception (a gated metric selected by `i_metric` still renders + // while locked) apply to them. { label: 'Measured Energy', labelZh: '实测能耗', - metrics: MEASURED_ENERGY_METRIC_CONFIG_KEYS, + metrics: [...MEASURED_ENERGY_METRIC_CONFIG_KEYS, ...POWER_BASIS_METRIC_CONFIG_KEYS], gated: true, }, { diff --git a/packages/app/src/components/inference/perf-ruler-store.test.ts b/packages/app/src/components/inference/perf-ruler-store.test.ts new file mode 100644 index 000000000..22e50d0e8 --- /dev/null +++ b/packages/app/src/components/inference/perf-ruler-store.test.ts @@ -0,0 +1,246 @@ +// @vitest-environment jsdom +import { act, createElement } from 'react'; +import { createRoot, type Root } from 'react-dom/client'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; + +import { track } from '@/lib/analytics'; +import { perfRulerAxisMetricKey } from '@/hooks/usePerfRulerAxisReset'; +import { + EMPTY_PERF_RULER_STATE, + nextPerfRulerState, + serializePerfRulers, +} from '@/lib/d3-chart/layers/perf-ruler'; + +import { + type PerfRulerStore, + PerfRulerStoreContext, + persistedPerfRulerAxisKey, + usePerfRulerStore, + usePerfRulerStoreValue, +} from './perf-ruler-store'; + +vi.mock('@/lib/analytics', () => ({ track: vi.fn() })); + +// `graphs` is always [interactivity, e2e]; ChartDisplay renders the e2e +// graph as chart-0 for every non-interactivity x mode. +const graphs = (e2eXField: string) => [ + { chartDefinition: { chartType: 'interactivity', x_scale_field: 'p90_intvty' } }, + { chartDefinition: { chartType: 'e2e', x_scale_field: e2eXField } }, +]; + +describe('persisted perf-ruler store', () => { + let container: HTMLDivElement; + let root: Root; + + const CURVE_A = 'roofline-b200_trt_fp8'; + const CURVE_B = 'roofline-mi355x_sglang_fp4'; + const OVERLAY = 'overlay-roofline-h100_vllm_fp8_run1'; + const AXIS = perfRulerAxisMetricKey('p90_e2el', 'y_totalTokensPerDollarTco'); + + interface Harness { + store: () => PerfRulerStore; + rerender: (axisMetricKey: string | null) => void; + } + + /** + * Hosts the store hook the way `InferenceProvider` does and reads it back + * through `usePerfRulerStore`, so the test observes what a chart mounted + * under the provider observes. + */ + function mountStore( + initialSerialized: string | undefined, + axisMetricKey: string | null, + ): Harness { + let latest!: PerfRulerStore; + let axisKey = axisMetricKey; + const Consumer = () => { + latest = usePerfRulerStore()!; + return null; + }; + function Host() { + const store = usePerfRulerStoreValue('chart-0', initialSerialized, axisKey); + return createElement( + PerfRulerStoreContext.Provider, + { value: store }, + createElement(Consumer), + ); + } + const render = () => act(() => root.render(createElement(Host))); + render(); + return { + store: () => latest, + rerender: (next) => { + axisKey = next; + render(); + }, + }; + } + + beforeEach(() => { + vi.mocked(track).mockClear(); + container = document.createElement('div'); + document.body.append(container); + root = createRoot(container); + }); + + afterEach(() => { + act(() => root.unmount()); + container.remove(); + }); + + it('is absent outside the provider so charts fall back to local state', () => { + let seen: PerfRulerStore | undefined | null = null; + const Consumer = () => { + seen = usePerfRulerStore(); + return null; + }; + act(() => root.render(createElement(Consumer))); + expect(seen).toBeUndefined(); + }); + + it('parses the share link into pending rulers and commits nothing until curves exist', () => { + const h = mountStore(`41.5|${CURVE_A}|${CURVE_B};120|${CURVE_A}|${OVERLAY}`, null); + expect(h.store().chartId).toBe('chart-0'); + expect(h.store().state).toBe(EMPTY_PERF_RULER_STATE); + expect(h.store().pending).toEqual([ + { id: 1, curveA: CURVE_A, curveB: CURVE_B, isoX: 41.5 }, + { id: 2, curveA: CURVE_A, curveB: OVERLAY, isoX: 120 }, + ]); + expect(serializePerfRulers(h.store().state)).toBe(''); + expect(track).not.toHaveBeenCalled(); + }); + + it('starts with no pending rulers when the link carries none or only garbage', () => { + expect(mountStore(undefined, null).store().pending).toBeNull(); + act(() => root.unmount()); + root = createRoot(container); + expect(mountStore('not-a-ruler', null).store().pending).toBeNull(); + }); + + it('commits resolved rulers with fresh ids, keeps the rest pending, and reports the link once', () => { + const h = mountStore(`41.5|${CURVE_A}|${CURVE_B};120|${CURVE_A}|${OVERLAY}`, AXIS); + // A ruler placed by hand before the link's curves resolved keeps its id. + act(() => { + h.store().setState((prev) => nextPerfRulerState(prev, { curve: 'x', isoX: 5 })); + h.store().setState((prev) => nextPerfRulerState(prev, { curve: 'y', isoX: 5 })); + }); + expect(h.store().state.rulers.map((ruler) => ruler.id)).toEqual([1]); + + const [official, overlay] = h.store().pending!; + act(() => h.store().commitPending([{ ...official, isoX: 42 }], [overlay])); + expect(h.store().state.rulers).toEqual([ + { id: 1, curveA: 'x', curveB: 'y', isoX: 5 }, + { id: 2, curveA: CURVE_A, curveB: CURVE_B, isoX: 42 }, + ]); + expect(h.store().state.nextId).toBe(3); + expect(h.store().pending).toEqual([overlay]); + // One opened link is one restore: the event carries the number of + // rulers the link held, even though the curves resolve in two passes. + expect(track).toHaveBeenCalledTimes(1); + expect(track).toHaveBeenCalledWith('interactivity_perf_ruler_shared_load', { count: 2 }); + expect(serializePerfRulers(h.store().state)).toBe(`5|x|y;42|${CURVE_A}|${CURVE_B}`); + + act(() => h.store().commitPending([overlay], null)); + expect(h.store().pending).toBeNull(); + expect(h.store().state.rulers.map((ruler) => ruler.curveB)).toEqual(['y', CURVE_B, OVERLAY]); + expect(track).toHaveBeenCalledTimes(1); + }); + + it('does not report a restore when only unmeasurable rulers were dropped', () => { + const h = mountStore(`41.5|${CURVE_A}|${CURVE_B}`, AXIS); + act(() => h.store().commitPending([], null)); + expect(h.store().pending).toBeNull(); + expect(h.store().state).toBe(EMPTY_PERF_RULER_STATE); + expect(track).not.toHaveBeenCalled(); + }); + + it('keeps share-link rulers through the first chart definition but resets on an axis change', () => { + const h = mountStore(`41.5|${CURVE_A}|${CURVE_B}`, null); + // The first data arriving (null → key) is not an axis change. + h.rerender(AXIS); + expect(h.store().pending).toHaveLength(1); + // Loading a new data set (key → null → same key) is not one either. + h.rerender(null); + h.rerender(AXIS); + expect(h.store().pending).toHaveLength(1); + + act(() => h.store().commitPending(h.store().pending!, null)); + act(() => h.store().setState((prev) => nextPerfRulerState(prev, { curve: 'x', isoX: 5 }))); + expect(h.store().state.rulers).toHaveLength(1); + expect(h.store().state.draft).not.toBeNull(); + + // Placed on the old axes: committed rulers, the draft, and anything still + // pending are gone once the y metric changes. + h.rerender(perfRulerAxisMetricKey('p90_e2el', 'y_tpPerGpu')); + expect(h.store().state.rulers).toEqual([]); + expect(h.store().state.draft).toBeNull(); + expect(h.store().state.nextId).toBe(2); + expect(h.store().pending).toBeNull(); + expect(serializePerfRulers(h.store().state)).toBe(''); + }); + + it('discards pending rulers on an axis change even before any commit', () => { + const h = mountStore(`41.5|${CURVE_A}|${CURVE_B}`, AXIS); + h.rerender(perfRulerAxisMetricKey('p90_ttft', 'y_totalTokensPerDollarTco')); + expect(h.store().pending).toBeNull(); + }); + + it('discardPending drops the link rulers without touching committed ones', () => { + const h = mountStore(`41.5|${CURVE_A}|${CURVE_B};9|${CURVE_A}|${OVERLAY}`, AXIS); + const [official, overlay] = h.store().pending!; + act(() => h.store().commitPending([official], [overlay])); + const committed = h.store().state; + act(() => h.store().discardPending()); + expect(h.store().pending).toBeNull(); + expect(h.store().state).toBe(committed); + }); + + describe('persistedPerfRulerAxisKey (axis identity of the rendered chart)', () => { + const Y = 'y_totalTokensPerDollarTco'; + + it('is null until a graph exists', () => { + expect(persistedPerfRulerAxisKey([], 'interactivity', Y)).toBeNull(); + }); + + it('follows the graph the x mode renders, not graphs[0]', () => { + const interactivity = persistedPerfRulerAxisKey(graphs('p90_e2el'), 'interactivity', Y); + const e2e = persistedPerfRulerAxisKey(graphs('p90_e2el'), 'e2e', Y); + const ttft = persistedPerfRulerAxisKey(graphs('p90_ttft'), 'ttft', Y); + expect(new Set([interactivity, e2e, ttft]).size).toBe(3); + // A stable mode and data set keep the identity, so rulers survive + // reloads, legend toggles, and date changes. + expect(persistedPerfRulerAxisKey(graphs('p90_e2el'), 'e2e', Y)).toBe(e2e); + }); + + it('changes with the percentile and the y metric', () => { + const base = persistedPerfRulerAxisKey(graphs('p90_e2el'), 'e2e', Y); + expect(persistedPerfRulerAxisKey(graphs('p99_e2el'), 'e2e', Y)).not.toBe(base); + expect(persistedPerfRulerAxisKey(graphs('p90_e2el'), 'e2e', 'y_tpPerGpu')).not.toBe(base); + }); + + it('treats a derived agentic x mode as its own axis even when the e2e field is unchanged', () => { + // ChartDisplay overrides the e2e graph's x field for derived modes + // locally; the provider's definition still says p90_e2el. + expect( + persistedPerfRulerAxisKey(graphs('p90_e2el'), 'e2e-normalized-interactivity', Y), + ).not.toBe(persistedPerfRulerAxisKey(graphs('p90_e2el'), 'e2e', Y)); + }); + + it('falls back to the first graph when no graph matches the mode', () => { + // Mirrors ChartDisplay's `visibleGraphs` fallback: with only an + // interactivity graph, chart-0 renders it in every mode. + const only = [graphs('p90_e2el')[0]]; + const key = persistedPerfRulerAxisKey(only, 'e2e', Y); + expect(key).not.toBeNull(); + expect(persistedPerfRulerAxisKey(only, 'e2e', 'y_tpPerGpu')).not.toBe(key); + }); + }); + + it('keeps the store identity stable across unrelated rerenders', () => { + const h = mountStore(`41.5|${CURVE_A}|${CURVE_B}`, AXIS); + const before = h.store(); + h.rerender(AXIS); + expect(h.store()).toBe(before); + expect(h.store().setState).toBe(before.setState); + }); +}); diff --git a/packages/app/src/components/inference/perf-ruler-store.ts b/packages/app/src/components/inference/perf-ruler-store.ts new file mode 100644 index 000000000..98bbfd692 --- /dev/null +++ b/packages/app/src/components/inference/perf-ruler-store.ts @@ -0,0 +1,179 @@ +'use client'; + +import { + type Dispatch, + type SetStateAction, + createContext, + useCallback, + useContext, + useMemo, + useRef, + useState, +} from 'react'; + +import { track } from '@/lib/analytics'; +import { perfRulerAxisMetricKey } from '@/hooks/usePerfRulerAxisReset'; +import { + EMPTY_PERF_RULER_STATE, + MAX_PERF_RULERS, + type PerfRulerMeasurement, + type PerfRulerState, + clearPerfRulers, + parsePerfRulers, +} from '@/lib/d3-chart/layers/perf-ruler'; + +/** + * @file perf-ruler-store.ts + * @description Provider-owned Perf Ruler state for the primary inference + * chart, so completed rulers persist in share links (`i_rulers`). Lives + * beside `InferenceContext` rather than inside it so `ScatterGraph` can + * consume the store without importing the (heavily mocked) provider module. + */ + +/** + * The chart instance whose Perf Rulers persist in share links. `ChartDisplay` + * mounts the primary chart as `chart-${graphIndex}` and only graph 0 is ever + * visible; the replay chart (`replay-chart-0`) draws interpolated frames of + * the same curves and must NOT bind, or every ruler would render twice and + * the replay's prune pass could delete rulers the main chart still shows. + */ +export const PERSISTED_PERF_RULER_CHART_ID = 'chart-0'; + +/** + * Perf-ruler store for the persisted chart. Lives in its own context rather + * than the Display domain so a ruler commit does not rerender every display + * consumer, and so harnesses that mount `InferenceContextsProvider` with + * static mock values (no store) keep the chart's component-local fallback. + * + * `state` holds COMMITTED rulers: the D3 layer renders them and `i_rulers` + * serializes them. `pending` holds rulers parsed from the share link whose + * curves may not have been drawn yet — data, `i_gpus`, comparison dates, + * and `?unofficialrun=` overlays all arrive after the chart's first draw, + * and the chart prunes any committed ruler whose curve path is absent from + * the DOM. The chart therefore commits a pending ruler only once BOTH of + * its curve paths exist (see the perf-ruler decoration effect in + * ScatterGraph); rulers whose curves never appear stay pending, invisible + * and unserialized, until an axis change or an explicit clear discards them. + */ +export interface PerfRulerStore { + chartId: string; + state: PerfRulerState; + setState: Dispatch>; + pending: readonly PerfRulerMeasurement[] | null; + /** + * Commit share-link rulers whose curves now exist (`resolved`, iso-x + * already clamped to the pair's overlap) and keep `remaining` pending. + */ + commitPending: ( + resolved: readonly PerfRulerMeasurement[], + remaining: readonly PerfRulerMeasurement[] | null, + ) => void; + /** Drop share-link rulers that were never committed (toggle-off, clear). */ + discardPending: () => void; +} + +/** + * Axis identity of the chart `ChartDisplay` renders as `chart-0`, for the + * store's axis reset. `graphs` is always `[interactivity, e2e]`, but + * ChartDisplay shows the e2e graph for every non-interactivity x mode + * (`visibleGraphs`), so the rendered chart — not `graphs[0]` — is what the + * rulers were placed on. The x mode itself is part of the identity as well: + * the derived agentic modes (e2e-normalized interactivity, …) override the + * e2e graph's `x_scale_field` inside ChartDisplay only, so here the same + * definition still reads `_e2el` for those modes. Percentile changes + * are already encoded in `x_scale_field`. Null while no graph exists. + */ +export function persistedPerfRulerAxisKey( + graphs: readonly { chartDefinition: { chartType: string; x_scale_field: string } }[], + xAxisMode: string, + yAxisMetric: string, +): string | null { + const wantedType = xAxisMode === 'interactivity' ? 'interactivity' : 'e2e'; + const graph = + graphs.find((candidate) => candidate.chartDefinition.chartType === wantedType) ?? graphs[0]; + if (!graph) return null; + return perfRulerAxisMetricKey(`${xAxisMode}:${graph.chartDefinition.x_scale_field}`, yAxisMetric); +} + +/** Provided by `InferenceProvider`; exported for chart component tests. */ +export const PerfRulerStoreContext = createContext(undefined); + +/** The persisted-ruler store, or undefined outside `InferenceProvider`. */ +export function usePerfRulerStore(): PerfRulerStore | undefined { + return useContext(PerfRulerStoreContext); +} + +/** + * Owns the persisted perf-ruler state. Exported so component tests can host a + * real store around a chart without the full provider. + * + * `axisMetricKey` is the persisted chart's axis identity + * ({@link persistedPerfRulerAxisKey}), or null while no chart definition exists. + * The axis reset runs HERE, not through `usePerfRulerAxisReset` in the chart: + * that hook adjusts state during the chart's render, which is only legal for + * the chart's own state — updating a provider's state from a child's render + * is a React error. Same semantics: a change of either axis metric clears + * committed rulers (redrawn curves would give a ratio nobody placed) and + * discards pending ones (they were placed on the old axes). The null → key + * transition on first data is not a change, so share-link rulers survive + * the load; the x-mode fallback for fixed sequences also settles before any + * chart definition exists. + */ +export function usePerfRulerStoreValue( + chartId: string, + initialSerialized: string | undefined, + axisMetricKey: string | null, +): PerfRulerStore { + const [state, setState] = useState(EMPTY_PERF_RULER_STATE); + const [pending, setPending] = useState(() => { + const parsed = parsePerfRulers(initialSerialized).rulers; + return parsed.length > 0 ? parsed : null; + }); + // `interactivity_perf_ruler_shared_load` fires once per store — once per + // opened link — with the number of rulers the link carried, the first time + // any of them renders. Rulers commit per curve arrival (below), so a + // per-commit event would count one link several times with partial counts. + const linkRulerCountRef = useRef(pending?.length ?? 0); + const sharedLoadReportedRef = useRef(false); + + const [appliedAxisMetricKey, setAppliedAxisMetricKey] = useState(axisMetricKey); + if (axisMetricKey !== null && axisMetricKey !== appliedAxisMetricKey) { + setAppliedAxisMetricKey(axisMetricKey); + if (appliedAxisMetricKey !== null) { + setState(clearPerfRulers); + setPending(null); + } + } + + const commitPending = useCallback( + ( + resolved: readonly PerfRulerMeasurement[], + remaining: readonly PerfRulerMeasurement[] | null, + ) => { + if (resolved.length > 0) { + setState((prev) => { + // Fresh ids from the live counter: a ruler placed by hand before the + // share-link rulers resolved must keep its own join key. + const rulers = [ + ...prev.rulers, + ...resolved.map((ruler, index) => ({ ...ruler, id: prev.nextId + index })), + ]; + while (rulers.length > MAX_PERF_RULERS) rulers.shift(); + return { rulers, draft: prev.draft, nextId: prev.nextId + resolved.length }; + }); + if (!sharedLoadReportedRef.current) { + sharedLoadReportedRef.current = true; + track('interactivity_perf_ruler_shared_load', { count: linkRulerCountRef.current }); + } + } + setPending(remaining); + }, + [], + ); + const discardPending = useCallback(() => setPending(null), []); + + return useMemo( + () => ({ chartId, state, setState, pending, commitPending, discardPending }), + [chartId, state, pending, commitPending, discardPending], + ); +} diff --git a/packages/app/src/components/inference/types.ts b/packages/app/src/components/inference/types.ts index 5a74eeb2c..b91759d2b 100644 --- a/packages/app/src/components/inference/types.ts +++ b/packages/app/src/components/inference/types.ts @@ -6,6 +6,7 @@ import type { Model, Sequence } from '@/lib/data-mappings'; import type { PowerTier } from '@/lib/power-tier'; import type { SystemPowerEstimate } from '@/lib/modeled-system-power'; import type { MetricKey } from './metric-registry'; +import type { PowerBasis } from '@/lib/power-basis'; export type { WorkerPower }; @@ -347,8 +348,67 @@ export interface InferenceData extends Partial void; setScaleType: (type: 'auto' | 'linear' | 'log') => void; + setPowerCompare: (mode: PowerCompare) => void; setQuickFilterVendors: (vendors: string[]) => void; setQuickFilterFrameworks: (frameworks: string[]) => void; setQuickFilterDeployment: (modes: DeploymentMode[]) => void; diff --git a/packages/app/src/components/inference/ui/CacheReuseLink.tsx b/packages/app/src/components/inference/ui/CacheReuseLink.tsx new file mode 100644 index 000000000..4c7c213ec --- /dev/null +++ b/packages/app/src/components/inference/ui/CacheReuseLink.tsx @@ -0,0 +1,36 @@ +'use client'; + +import Link from 'next/link'; + +import type { PointMeta } from '@/hooks/api/use-trace-server-metrics'; +import { track } from '@/lib/analytics'; +import { cacheReuseHref } from '@/lib/cache-reuse-link'; +import { useLocale } from '@/lib/use-locale'; + +const STRINGS = { + en: { chart: 'Prefix cache reuse →', point: 'Prefix cache reuse for this config →' }, + zh: { chart: '前缀缓存复用 →', point: '查看此配置的前缀缓存复用 →' }, +} as const; + +/** Agentic-only link into the Prefix Cache Reuse tab, from the chart footer or a point. */ +export function CacheReuseLink({ point, className }: { point?: PointMeta; className?: string }) { + const locale = useLocale(); + const t = STRINGS[locale]; + return ( + + track('inference_cache_reuse_link_clicked', { + from: point ? 'point' : 'chart', + ...(point ? { id: point.id } : {}), + }) + } + > + {point ? t.point : t.chart} + + ); +} diff --git a/packages/app/src/components/inference/ui/ChartControls.tsx b/packages/app/src/components/inference/ui/ChartControls.tsx index 1f896701a..726a6918c 100644 --- a/packages/app/src/components/inference/ui/ChartControls.tsx +++ b/packages/app/src/components/inference/ui/ChartControls.tsx @@ -61,6 +61,7 @@ import { MetricExplanation } from './MetricExplanation'; import { PowerMetricAvailability } from './PowerMetricAvailability'; import { MeasuredMetricControls } from './MeasuredMetricControls'; import { + changeMeasuredMetricConfig, getMeasuredMetricConfig, MEASURED_METRIC_DEFAULTS, type MeasuredMetricFamily, @@ -96,7 +97,7 @@ const STRINGS = { gpuConfig: 'Chip Config', gpuConfigTooltip: 'Select up to 4 chip configurations to compare their historical performance over time. This allows for tracking how software updates may affect specific hardware.', - gpuConfigPlaceholder: 'Select a Chip Config for comparison', + gpuConfigPlaceholder: 'Select Chip Config', comparisonDateRange: 'Comparison Date Range', comparisonDateRangeTooltip: 'Select the start and end dates for the historical comparison. The chart will show performance data for the selected chip configs across this time range.', @@ -140,7 +141,7 @@ const STRINGS = { gpuConfig: '芯片配置', gpuConfigTooltip: '最多选择 4 个芯片配置以对比其历史性能趋势。可用于追踪软件更新对特定硬件的影响。', - gpuConfigPlaceholder: '选择芯片配置进行对比', + gpuConfigPlaceholder: '选择芯片配置', comparisonDateRange: '对比日期范围', comparisonDateRangeTooltip: '选择历史对比的起止日期。图表将展示所选芯片配置在此时间范围内的性能数据。', @@ -230,6 +231,7 @@ export default function ChartControls({ selectedXAxisMetric, selectedXAxisMode, scaleType, + powerCompare, } = useInferenceDisplay(); const { setSelectedModel, @@ -242,6 +244,7 @@ export default function ChartControls({ setSelectedDateRange, setSelectedXAxisMetric, setScaleType, + setPowerCompare, } = useInferenceActions(); // Y-axis options come from the canonical registry and need no API data. @@ -335,10 +338,14 @@ export default function ChartControls({ if (!config) return [option]; if (seen.has(config.family)) return []; seen.add(config.family); + // Keep the selected boundary (and other dimensions) when hopping between + // the power and energy families; fall back to the family default otherwise. const value = selectedConfig?.family === config.family ? selectedYAxisMetric - : MEASURED_METRIC_DEFAULTS[config.family]; + : selectedConfig + ? changeMeasuredMetricConfig(selectedYAxisMetric, { family: config.family }) + : MEASURED_METRIC_DEFAULTS[config.family]; return [ { value, @@ -571,6 +578,8 @@ export default function ChartControls({
`vs. ${word} Time To First Token`, vsE2eLatency: (pctl?: string) => pctl ? `vs. ${pctl} End-to-end Latency` : 'vs. End-to-end Latency', @@ -165,6 +181,15 @@ const STRINGS = { noChartData: '当前模型、场景与筛选条件下没有匹配的基准测试数据。请调整上方筛选条件查看结果。', noSystemPowerData: '当前选择没有可用的系统功耗估算。请选择 8K / 1K 场景;估算仅覆盖 GPU 遥测已验证、硬件受支持、八卡机箱位置已知的运行。存在遥测数据时,仍可单独查看 GPU 实测功耗。', + noUtilityModeledData: + '当前选择没有可用的数据中心建模数值。该边界需要 8K / 1K 场景、已验证的 GPU 遥测,且硬件在机箱功耗模型覆盖范围内(不含 NVL72 系统)。可切换到其他功耗边界以保留数据点。', + powerBasisAssumptions: { + 'gpu-provisioned': + 'GPU 额定边界 · 功率取硬件注册表中每 GPU 的额定 TDP,因此每种硬件的功率曲线为水平线。每输出 token 能耗 = TDP × 分配的 GPU 数 ÷ 整个部署的输出 tok/s;分离式配置将 prefill 与 decode GPU 一并计入。未公布 TDP 的硬件不绘制。', + 'utility-provisioned': + '全电源配置边界 · 功率取硬件注册表中每 GPU 的全电源配置(all-in)市电功率(来源:SemiAnalysis Datacenter Industry Model),因此每种硬件的功率曲线为水平线。每输出 token 能耗 = all-in 功率 × 分配的 GPU 数 ÷ 整个部署的输出 tok/s;分离式配置将 prefill 与 decode GPU 一并计入,这与未加门控的“每输出 token 全电源配置能耗”按 decode GPU 计算不同。', + 'utility-modeled': `数据中心建模边界 · 将 GPU 实测功耗经机箱功耗模型(CPU、DRAM、平台开销、PSU 损耗)推算至市电侧:机箱交流功耗估算 × PUE ${AIR_COOLED_SYSTEM_PUE}(风冷,仅应用一次),再除以实测 GPU 数;每输出 token 能耗按同一比例放大实测能耗。机箱功耗模型版本 ${SYSTEM_POWER_MODEL_REVISION.slice(0, 7)}。仅适用于 8K / 1K、遥测已验证且硬件受支持的运行;NVL72 系统(GB200、GB300)及缺少数值的数据点不绘制。`, + }, vsTtft: (word: string) => `vs. ${word === 'Median' ? '中位' : word} 首 token 延迟(TTFT)`, vsE2eLatency: (pctl?: string) => (pctl ? `vs. ${pctl} 端到端延迟` : 'vs. 端到端延迟'), }, @@ -290,9 +315,19 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean selectedXAxisMode, tokenRevenuePricing, showLineLabels, + powerCompare, } = useInferenceDisplay(); const { setSelectedDates, setSelectedDatesFromRunExpansion, setIsLegendExpanded } = useInferenceActions(); + // The metric key carries the power boundary; the caption discloses it for + // the derived boundaries (there is no separate URL param). + const selectedPowerBasis = getMeasuredMetricConfig(selectedYAxisMetric)?.basis; + // The Measured Power "Timeline" display swaps the scatter body for the + // per-second telemetry traces (PowerTimeline); table view and captions are + // unchanged because the metric key aliases the measured average. + const selectedMeasuredConfig = getMeasuredMetricConfig(selectedYAxisMetric); + const isPowerTimeline = + selectedMeasuredConfig?.family === 'power' && selectedMeasuredConfig.display === 'timeline'; const selectedBenchmarkType: 'single_turn' | 'agentic_traces' = selectedSequence === Sequence.AgenticTraces ? 'agentic_traces' : 'single_turn'; const workflowInfoBenchmarkType = @@ -455,6 +490,7 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean selectedPercentile, tcoBasis, selectedXAxisMode, + powerCompare, }, ); @@ -504,6 +540,7 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean selectedXAxisMetric, selectedE2eXAxisMetric, selectedPercentile, + powerCompare, selectedXAxisMode, tokenRevenuePricing, tcoBasis, @@ -824,7 +861,9 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean

{isModeledSystemPowerConfigKey(selectedYAxisMetric) ? t.noSystemPowerData - : t.noChartData} + : selectedPowerBasis === 'utility-modeled' + ? t.noUtilityModeledData + : t.noChartData}

, ] @@ -883,6 +922,7 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean {hasOffloadHalo && } {hasLegacyPowerPoints && } {isAgenticSequence && } + {isAgenticSequence && !minimalChrome && } {hasAtomSeries && ( )} @@ -1011,6 +1051,8 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean {getSequenceLabel(graph.sequence as Sequence, locale)}{' '} {metricChartTitle(graph.chartDefinition, selectedYAxisMetric, locale)}{' '} {(() => { + // The timeline's x axis is time, not the scatter x metric. + if (isPowerTimeline) return null; const xField = graph.chartDefinition.x_scale_field; if (xField?.endsWith('_ttft')) { const percentile = xField.replace(/_ttft$/u, ''); @@ -1138,6 +1180,15 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean {t.systemPowerAssumptions}

)} + {selectedPowerBasis && selectedPowerBasis !== 'gpu-measured' && ( +

+ {t.powerBasisAssumptions[selectedPowerBasis]} +

+ )} {isUnofficialRun && selectedXAxisMode === 'e2e-normalized-interactivity' && (

@@ -1187,11 +1238,42 @@ export default function ChartDisplay({ embedded = false }: { embedded?: boolean ); } + if (isPowerTimeline) { + return ( +

+ entry.point), + ]} + overlayData={ + selectUnofficialOverlayForMode( + selectedXAxisMode, + graph.chartDefinition.chartType, + overlayDataByChartType, + ) ?? undefined + } + yLabel={metricLabel( + graph.chartDefinition, + selectedYAxisMetric, + locale, + )} + caption={chartCaption} + /> +
+ ); + } + return isGpuComparison ? ( !point.powerVariant)} xLabel={resolvedXLabel} yLabel={metricLabel(graph.chartDefinition, selectedYAxisMetric, locale)} chartDefinition={graph.chartDefinition} diff --git a/packages/app/src/components/inference/ui/GPUGraph.tsx b/packages/app/src/components/inference/ui/GPUGraph.tsx index 01fc782eb..f5cb0c7e4 100644 --- a/packages/app/src/components/inference/ui/GPUGraph.tsx +++ b/packages/app/src/components/inference/ui/GPUGraph.tsx @@ -48,6 +48,7 @@ import { chartFrontier, upperPowerEnvelope, isPowerCurveMetric, + isPowerGaugeSeries, isMeasuredPowerCurveMetric, } from '@/components/inference/utils/powerCurves'; import type { @@ -443,10 +444,20 @@ const GPUGraph = React.memo( if (!powerEnvelopeMode) return paretoRooflines; const result: Record = {}; for (const [key, points] of Object.entries(groupedData)) { - result[key] = upperPowerEnvelope(points, chartDefinition.chartType !== 'e2e'); + result[key] = upperPowerEnvelope( + points, + chartDefinition.chartType !== 'e2e', + isPowerGaugeSeries(selectedYAxisMetric, points[0]), + ); } return result; - }, [powerEnvelopeMode, groupedData, paretoRooflines, chartDefinition.chartType]); + }, [ + powerEnvelopeMode, + groupedData, + paretoRooflines, + chartDefinition.chartType, + selectedYAxisMetric, + ]); const boundaryPointKeys = useMemo(() => { const keys = new Set(); diff --git a/packages/app/src/components/inference/ui/InferenceTable.tsx b/packages/app/src/components/inference/ui/InferenceTable.tsx index b6cdafa7e..3e6ce68f9 100644 --- a/packages/app/src/components/inference/ui/InferenceTable.tsx +++ b/packages/app/src/components/inference/ui/InferenceTable.tsx @@ -7,6 +7,7 @@ import { type DataTableColumn, DataTable } from '@/components/ui/data-table'; import { chipCounts } from '@/lib/chip-counts'; import { getNestedYValue, metricLabel, xAxisLabel } from '@/lib/chart-utils'; import { isModeledSystemPowerConfigKey } from '@/components/inference/metric-registry'; +import { inferPowerCompare, powerSeriesLabel } from '@/components/inference/utils/power-compare'; import { sortRowsByYMetric } from '@/components/inference/ui/inference-table-sort'; import { type Precision, getPrecisionLabel } from '@/lib/data-mappings'; import { getDisplayLabel } from '@/lib/utils'; @@ -43,6 +44,7 @@ export function inferenceTableHeaderLabels( physicalChips: locale === 'zh' ? '物理芯片数' : 'Physical Chips', configuredChips: locale === 'zh' ? '配置中的芯片数' : 'Configured Chip Count', concurrency: locale === 'zh' ? '并发数' : 'Conc', + series: locale === 'zh' ? '系列' : 'Series', yMetric: metricLabel(chartDefinition, selectedYAxisMetric, locale), xMetric: xAxisLabel(chartDefinition, locale), throughput: locale === 'zh' ? '单芯片吞吐量 (tok/s)' : 'Throughput/Chip (tok/s)', @@ -66,6 +68,9 @@ export default function InferenceTable({ () => sortRowsByYMetric(data, chartDefinition, selectedYAxisMetric), [data, chartDefinition, selectedYAxisMetric], ); + // Boundary / role clones (`i_pcompare`) share every config column with their + // base row; the series column is what tells them apart. + const powerCompare = useMemo(() => inferPowerCompare(data), [data]); const columns = useMemo[]>( () => [ @@ -85,6 +90,19 @@ export default function InferenceTable({ className: 'whitespace-nowrap', importance: 'key', }, + ...(powerCompare === 'none' + ? [] + : [ + { + header: headers.series, + cell: (row: InferenceData) => + powerSeriesLabel(row, selectedYAxisMetric, powerCompare, locale), + sortValue: (row: InferenceData) => + powerSeriesLabel(row, selectedYAxisMetric, powerCompare, locale), + className: 'whitespace-nowrap', + importance: 'key' as const, + }, + ]), { header: headers.tensorParallelism, align: 'right', @@ -151,7 +169,7 @@ export default function InferenceTable({ importance: 'key', }, ], - [yPath, headers, showModeledPower], + [yPath, headers, showModeledPower, powerCompare, selectedYAxisMetric, locale], ); return ( diff --git a/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx b/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx index 46f0600eb..da5068e02 100644 --- a/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx +++ b/packages/app/src/components/inference/ui/MeasuredMetricControls.tsx @@ -7,15 +7,24 @@ import { SelectTrigger, SelectValue, } from '@/components/ui/select'; +import { track } from '@/lib/analytics'; +import { POWER_BASES, POWER_BASIS_LABELS, type PowerBasis } from '@/lib/power-basis'; import { useLocale } from '@/lib/use-locale'; import { changeMeasuredMetricConfig, getMeasuredMetricConfig, type MeasuredMetricConfigChange, } from '../measured-metric-config'; +import type { PowerCompare } from '../types'; +import { POWER_COMPARE_MODES, powerCompareAvailable } from '../utils/power-compare'; const STRINGS = { en: { + basis: 'Boundary', + basisHelp: + 'Where power is counted. GPU measured: runner telemetry from the GPU boards. GPU provisioned: rated TDP per GPU. Utility provisioned: all-in provisioned utility power per GPU. Utility modeled: measured GPU power carried through the modeled chassis to the utility meter with PUE. Points without a value for the chosen boundary are omitted, never replaced with an estimate.', + basisHint: + 'Derived boundaries report whole-deployment average power and joules per output token. Changing another setting returns to GPU measured.', scope: 'Scope', scopeHelp: 'All GPUs measures the whole deployment. Prefill and decode select only GPUs serving that role.', @@ -29,7 +38,8 @@ const STRINGS = { roleHint: 'Prefill and decode power support Average only.', display: 'Display', displayHelp: - 'Power per chip in watts, or average power as a percentage of chip TDP. Percent of TDP is available for the all-GPU average only.', + 'Power per chip in watts, average power as a percentage of chip TDP, or the per-second telemetry timeline behind the average. Percent of TDP and Timeline are available for the all-GPU average only.', + timeline: 'Timeline', denominator: 'Per', denominatorHelp: 'Choose the energy denominator. All-GPU energy per input or output token includes the whole deployment; role energy is selected separately under Scope.', @@ -40,8 +50,20 @@ const STRINGS = { unit: 'Unit', unitHelp: 'Energy is shown in joules. Energy per successful query can also be shown in watt-hours.', + compare: 'Compare', + compareHelp: + 'Overlay sibling series on the same points, in the hardware colour with a dash per series. All boundaries: GPU measured, GPU provisioned, utility provisioned and utility modeled. Prefill vs decode: each worker pool next to the whole deployment; on the energy axis the prefill pool is carried onto the output-token axis by the served input:output ratio. Available for the whole-deployment average W/chip and J per output token.', + compareNone: 'Off', + compareBoundaries: 'All boundaries', + compareRoles: 'Prefill vs decode', + compareUnavailable: + 'The comparison is paused for this setting: it needs the whole-deployment average W/chip or J per output token.', }, zh: { + basis: '功耗边界', + basisHelp: + '选择功耗的计量边界。GPU 实测:来自 GPU 板卡的运行器遥测;GPU 额定:每 GPU 的额定 TDP;全电源配置:每 GPU 的全电源配置(all-in)市电功率;数据中心建模:将 GPU 实测功耗经机箱功耗模型推算至市电侧并计入 PUE。所选边界缺少数值的数据点将被省略,不会用估算值替代。', + basisHint: '推导边界仅提供整个部署的平均功耗和每输出 token 能耗;更改其他设置将返回 GPU 实测。', scope: '统计范围', scopeHelp: '全部 GPU 对应整个部署;预填充和解码仅统计承担相应任务的 GPU。', all: '全部 GPU', @@ -54,7 +76,8 @@ const STRINGS = { roleHint: '预填充和解码功率仅支持平均值。', display: '显示方式', displayHelp: - '显示单芯片功率(瓦),或平均功率占芯片 TDP 的百分比。TDP 百分比仅支持全部 GPU 的平均功率。', + '显示单芯片功率(瓦)、平均功率占芯片 TDP 的百分比,或平均值背后的逐秒遥测时间线。TDP 百分比和时间线仅支持全部 GPU 的平均功率。', + timeline: '时间线', denominator: '能耗分母', denominatorHelp: '选择能耗的分母。按输入或输出 token 归一化的全部 GPU 能耗仍包含整个部署;预填充或解码能耗需在统计范围中单独选择。', @@ -64,22 +87,44 @@ const STRINGS = { query: '成功请求', unit: '单位', unitHelp: '能耗以焦耳显示;每个成功请求的能耗也可显示为瓦时。', + compare: '对比', + compareHelp: + '在同一批数据点上叠加同源系列:颜色仍按硬件区分,每个系列用不同虚线表示。全部边界:GPU 实测、GPU 额定、全电源配置、数据中心建模;预填充 vs 解码:各 worker 池与整个部署并列,能耗轴上的预填充能耗按实际服务的输入/输出 token 比折算到每输出 token。仅适用于整个部署的平均 W/芯片和每输出 token 能耗。', + compareNone: '关闭', + compareBoundaries: '全部边界', + compareRoles: '预填充 vs 解码', + compareUnavailable: '当前设置下对比已暂停:需要整个部署的平均 W/芯片或每输出 token 能耗。', }, } as const; export function MeasuredMetricControls({ metric, onChange, + compare = 'none', + onCompareChange, }: { metric: string; onChange: (metric: string) => void; + /** Comparison series overlaid on the metric (`i_pcompare`). */ + compare?: PowerCompare; + onCompareChange?: (mode: PowerCompare) => void; }) { - const t = STRINGS[useLocale()]; + const locale = useLocale(); + const t = STRINGS[locale]; const config = getMeasuredMetricConfig(metric); if (!config) return null; + const compareLabels: Record = { + none: t.compareNone, + boundaries: t.compareBoundaries, + roles: t.compareRoles, + }; + const compareActive = compare !== 'none'; + const compareApplies = powerCompareAvailable(metric, compare); const change = (next: MeasuredMetricConfigChange) => onChange(changeMeasuredMetricConfig(metric, next)); + const basisId = `measured-${config.family}-basis`; const scopeId = `measured-${config.family}-scope`; + const derivedBasis = config.basis !== 'gpu-measured'; const roleScope = config.family === 'energy' ? config.denominator === 'input' @@ -91,9 +136,31 @@ export function MeasuredMetricControls({ return (
+
+ + +
{config.family === 'energy' && (
change({ statistic })} @@ -204,6 +271,13 @@ export function MeasuredMetricControls({ > % TDP + + {t.timeline} +
@@ -237,6 +311,59 @@ export function MeasuredMetricControls({
)} + {onCompareChange && ( +
+ + +
+ )} + {derivedBasis && ( +

+ {t.basisHint} +

+ )} + {compareActive && !compareApplies && ( +

+ {t.compareUnavailable} +

+ )}
); } diff --git a/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx b/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx index a3dd548b0..cad29732b 100644 --- a/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx +++ b/packages/app/src/components/inference/ui/PowerMetricAvailability.tsx @@ -2,8 +2,16 @@ import { useMemo } from 'react'; import { useInferenceData, useInferenceFilters } from '../InferenceContext'; -import { isMeasuredEnergyConfigKey, metricOptionTitle, type MetricKey } from '../metric-registry'; +import { + isMeasuredEnergyConfigKey, + isPowerBasisConfigKey, + metricOptionTitle, + POWER_BASIS_METRIC_CONFIG_KEYS, + type MetricKey, +} from '../metric-registry'; +import { getMeasuredMetricConfig } from '../measured-metric-config'; import type { InferenceData } from '../types'; +import { powerBasisNormalization } from '@/lib/power-basis'; import { matchesQuickFilters } from '../utils/quickFilters'; import { powerMetricAvailability, @@ -33,6 +41,17 @@ const STRINGS = { ambiguous: 'Whole-deployment energy schema unavailable', missing: 'Metric not reported', }, + basisLabels: { + available: 'Value available', + noSpec: 'No published spec for this hardware', + noThroughput: 'No output throughput reported', + noNormalization: 'Whole-deployment GPU count unavailable for this disaggregated row', + noTelemetry: 'No validated GPU telemetry', + invalid: 'Validation failed', + modelWorkload: 'Chassis model covers 8K / 1K only', + modelHardware: 'Hardware not in the chassis power model', + modelUnsupported: 'Chassis model unsupported for this deployment', + }, note: 'A missing verdict does not establish age or validity. Prefill/decode metrics measure separate worker pools. Missing values are never replaced with zero or TDP estimates.', all: 'Availability of all measured metrics', evidence: 'Selected metric: source details', @@ -54,6 +73,17 @@ const STRINGS = { ambiguous: '缺少整个部署的能耗 schema', missing: '未提供此指标', }, + basisLabels: { + available: '有数值', + noSpec: '该硬件没有公开的规格参数', + noThroughput: '未报告输出吞吐量', + noNormalization: '无法确定该分离式部署的 GPU 总数', + noTelemetry: '没有已验证的 GPU 遥测', + invalid: '验证失败', + modelWorkload: '机箱功耗模型仅覆盖 8K / 1K', + modelHardware: '硬件不在机箱功耗模型范围内', + modelUnsupported: '机箱功耗模型不支持此部署', + }, note: '缺少验证结论不能判断数据新旧或有效性。prefill/decode 指标仅衡量独立 worker 池。缺失值不会被替换为零或 TDP 估算值。', all: '所有实测指标的可用性', evidence: '当前指标的来源详情', @@ -61,6 +91,99 @@ const STRINGS = { }, } as const; +/** + * Why a point lacks a derived power boundary (lib/power-basis.ts). Provisioned + * boundaries are spec constants, so their watts exist for any registered + * hardware and their energy additionally needs output throughput plus, for + * disaggregated rows, the whole-deployment GPU count that + * `powerBasisNormalization` recovers only for fixed-sequence runs with integer + * prefill/decode counts. The modeled boundary is measured telemetry carried + * through the chassis model, so it inherits the telemetry verdict and the + * model's own unsupported reasons. + */ +export type PowerBasisAvailabilityState = + | 'available' + | 'noSpec' + | 'noThroughput' + | 'noNormalization' + | 'noTelemetry' + | 'invalid' + | 'modelWorkload' + | 'modelHardware' + | 'modelUnsupported'; + +const POWER_BASIS_AVAILABILITY_STATES: readonly PowerBasisAvailabilityState[] = [ + 'available', + 'noSpec', + 'noThroughput', + 'noNormalization', + 'noTelemetry', + 'invalid', + 'modelWorkload', + 'modelHardware', + 'modelUnsupported', +]; + +const hasFiniteValue = (point: InferenceData, key: MetricKey): boolean => { + const value = point[key]; + return ( + typeof value === 'object' && + value !== null && + 'y' in value && + typeof value.y === 'number' && + Number.isFinite(value.y) + ); +}; + +export function powerBasisState( + point: InferenceData, + configKey: string, +): PowerBasisAvailabilityState { + const key = configKey.replace(/^y_/u, '') as MetricKey; + if (hasFiniteValue(point, key)) return 'available'; + const config = getMeasuredMetricConfig(configKey); + if (config?.basis === 'utility-modeled') { + if (point.power_valid === 0) return 'invalid'; + const model = point.modeledSystemPower; + if (model?.status === 'unsupported') { + if (model.reason === 'workload') return 'modelWorkload'; + if (model.reason === 'hardware') return 'modelHardware'; + if (model.reason === 'telemetry') return 'noTelemetry'; + return 'modelUnsupported'; + } + // A supported model without a plotted value means B1 is absent (B4 follows B1). + return 'noTelemetry'; + } + // Provisioned energy needs the watts sibling plus the whole-deployment + // normalization; the same helper that withheld the value says which half is + // missing, so the explanation cannot drift from the formula. + const wattsKey = ( + config?.basis === 'gpu-provisioned' ? 'gpuProvisionedWatts' : 'utilityProvisionedWatts' + ) satisfies MetricKey; + if (!hasFiniteValue(point, wattsKey)) return 'noSpec'; + const perGpu = point.output_tput_per_gpu; + if (typeof perGpu !== 'number' || !Number.isFinite(perGpu) || perGpu <= 0) return 'noThroughput'; + // Throughput exists, so only the disaggregated GPU count can be missing. + // Chart points may lack the counts an aggregate entry always has; an + // unknown count is exactly the "unavailable" case the helper reports. + const { allocatedGpus } = powerBasisNormalization({ + output_tput_per_gpu: perGpu, + disagg: point.disagg ?? false, + benchmark_type: point.benchmark_type, + num_prefill_gpu: point.num_prefill_gpu ?? Number.NaN, + num_decode_gpu: point.num_decode_gpu ?? Number.NaN, + }); + return allocatedGpus === null ? 'noNormalization' : 'noThroughput'; +} + +function powerBasisAvailability(points: readonly InferenceData[], metric: string) { + const counts = Object.fromEntries( + POWER_BASIS_AVAILABILITY_STATES.map((state) => [state, 0]), + ) as Record; + for (const point of points) counts[powerBasisState(point, metric)]++; + return { metric, counts, available: counts.available, total: points.length }; +} + export function PowerMetricAvailabilityPanel({ points, metric, @@ -74,16 +197,35 @@ export function PowerMetricAvailabilityPanel({ }) { const locale = useLocale(); const t = STRINGS[locale]; - const availability = useMemo(() => powerMetricAvailability(points), [points]); + const availability = useMemo( + () => [ + ...powerMetricAvailability(points), + ...POWER_BASIS_METRIC_CONFIG_KEYS.map((key) => powerBasisAvailability(points, key)), + ], + [points], + ); const selected = availability.find((entry) => entry.metric === metric); if (!selected) return null; + const isBasis = isPowerBasisConfigKey(metric); + const stateOf = (point: InferenceData) => + isBasis ? powerBasisState(point, metric) : powerMetricState(point, metric); + // The two dictionaries overlap on `invalid`; the selected metric, not the + // key, decides which copy applies so the measured strings stay untouched. + const labelOf = (state: PowerAvailabilityState | PowerBasisAvailabilityState) => + isBasis + ? t.basisLabels[state as PowerBasisAvailabilityState] + : t.labels[state as PowerAvailabilityState]; const sources = new Map< string, - { point: InferenceData; state: PowerAvailabilityState; count: number } + { + point: InferenceData; + state: PowerAvailabilityState | PowerBasisAvailabilityState; + count: number; + } >(); for (const point of points) { - const state = powerMetricState(point, metric); - if (state === 'strict') continue; + const state = stateOf(point); + if (state === 'strict' || state === 'available') continue; const key = JSON.stringify([point.hwKey, point.run_url, state, point.power_invalid_reasons]); const group = sources.get(key); if (group) group.count++; @@ -108,7 +250,10 @@ export function PowerMetricAvailabilityPanel({

{Object.entries(selected.counts) .filter(([, count]) => count > 0) - .map(([state, count]) => `${t.labels[state as PowerAvailabilityState]}: ${count}`) + .map( + ([state, count]) => + `${labelOf(state as PowerAvailabilityState | PowerBasisAvailabilityState)}: ${count}`, + ) .join(' · ')}

{[...sources.values()].map(({ point, state, count }, index) => (
  • - {point.hwKey}: {t.labels[state]} ({count}) + {point.hwKey}: {labelOf(state)} ({count}) {point.power_invalid_reasons?.length ? ` · ${point.power_invalid_reasons.join(', ')}` : ''} @@ -221,7 +366,7 @@ export function PowerMetricAvailability({ quickFilters, compareGpuPair, ]); - if (!isMeasuredEnergyConfigKey(metric)) return null; + if (!isMeasuredEnergyConfigKey(metric) && !isPowerBasisConfigKey(metric)) return null; return ( ` end labels would only overlap. */ +const MAX_LABELED_TRACES = 40; +/** Up to this many undrawn configs are named individually; beyond, per hardware. */ +const MAX_LISTED_MISSING = 8; +const CHART_HEIGHT = 600; +const MARGIN = { top: 24, right: 84, bottom: 60, left: 64 }; + +const STRINGS = { + en: { + timeAxis: 'Time axis', + wall: 'Wall clock (UTC)', + elapsed: 'Since start', + xWall: 'Time (UTC)', + xElapsed: 'Time since telemetry start (m:ss)', + perGpu: 'One line per GPU', + perGpuHelp: 'Draw every GPU of a config instead of the mean across its GPUs.', + pools: 'Prefill / decode pools', + poolsHelp: + 'One line per worker-role pool: the summed board power of the prefill GPUs and of the decode GPUs of a config. Dashed references are pool size × rated TDP.', + utilityLines: 'All-in provisioned lines', + utilityHelp: + 'Dashed reference at the all-in provisioned utility power per GPU from the hardware registry (SemiAnalysis Datacenter Industry Model). Off by default because it compresses the traces.', + loading: (runs: number) => + `Loading GPU telemetry for ${runs} run${runs === 1 ? '' : 's'}… (GitHub artifacts, may take a minute)`, + loadError: (runId: string, message: string) => `Run ${runId}: ${message}`, + missing: (missing: number, total: number) => + `${missing} of ${total} measured configs have no telemetry trace and are not drawn.`, + missingReason: { + 'no-source': (count: number) => + `${count} predate per-config telemetry provenance in the benchmark row`, + 'no-run': (count: number) => `${count} carry no workflow run`, + 'run-not-fetched': (count: number) => `${count} come from runs that were not loaded`, + 'not-in-run': (count: number) => + `${count} have no gpu_metrics artifact or power-audit bundle in their run (expired, or another collector)`, + } satisfies Record string>, + missingUndrawn: 'Not drawn', + noTraces: + 'No telemetry traces for the visible hardware. Enable a series in the legend or choose another date.', + noArtifacts: + 'These points predate per-config telemetry artifacts, so no timeline is available for them.', + droppedRuns: (runs: number) => + `Telemetry from ${runs} more run${runs === 1 ? '' : 's'} was not loaded (limit ${POWER_TIMELINE_MAX_RUNS} runs per chart).`, + telemetry: 'Telemetry', + method: + 'One-second means of per-GPU board power (nvidia-smi / amd-smi, or DCGM on Slurm / Dynamo runs) over the whole benchmark job; the emphasized segment is the validated window behind the measured average. Dashed lines: rated TDP per hardware from the hardware registry.', + methodPools: + 'In pool mode each line is the summed power of one worker-role pool (prefill or decode GPUs) and the dashed references are pool size × rated TDP.', + instructions: + 'Shift+Scroll to zoom horizontally · Drag to pan · Double-click to reset · Click a point to pin tooltip', + dismiss: 'Click elsewhere to dismiss', + phase: { + before: 'Before window (startup / warmup)', + window: 'Measurement window', + after: 'After window', + unknown: 'Window not recorded', + } satisfies Record, + meanPerGpu: 'Mean per GPU', + gpus: (count: number) => `${count} GPU${count === 1 ? '' : 's'}`, + min: 'min', + max: 'max', + validated: 'Validated average', + sinceStart: 'since start', + tdp: 'TDP', + allIn: 'all-in', + poolShort: { prefill: 'prefill', decode: 'decode', all: 'all GPUs' } satisfies Record< + PowerPoolRole, + string + >, + yPool: 'GPU pool power (W)', + pool: 'Pool', + poolPower: 'Pool power', + poolTdp: 'pool TDP', + focused: (label: string) => `Focused on ${label}`, + showAll: 'Show all', + unofficialRun: 'Unofficial run', + branch: 'Branch', + viewWorkflow: 'View workflow run', + }, + zh: { + timeAxis: '时间轴', + wall: '实际时刻(UTC)', + elapsed: '相对起点', + xWall: '时间(UTC)', + xElapsed: '距遥测开始的时间(分:秒)', + perGpu: '每个 GPU 一条线', + perGpuHelp: '绘制配置中每个 GPU 的曲线,而不是各 GPU 的平均值。', + pools: '预填充 / 解码 GPU 池', + poolsHelp: + '按 worker 角色分池绘制:每条线是同一配置中预填充 GPU 或解码 GPU 的板卡功耗之和。虚线参考为池内 GPU 数量 × 额定 TDP。', + utilityLines: '全电源配置参考线', + utilityHelp: + '按硬件注册表中每 GPU 的全电源配置(all-in)市电功率绘制虚线参考(SemiAnalysis 数据中心行业模型)。默认关闭,因为它会压缩曲线的纵向分辨率。', + loading: (runs: number) => + `正在加载 ${runs} 个运行的 GPU 遥测数据……(GitHub 产物,可能需要约一分钟)`, + loadError: (runId: string, message: string) => `运行 ${runId}:${message}`, + missing: (missing: number, total: number) => + `${total} 个有实测值的配置中有 ${missing} 个没有遥测曲线,未绘制。`, + missingReason: { + 'no-source': (count: number) => `${count} 个的基准测试行早于按配置记录的遥测来源`, + 'no-run': (count: number) => `${count} 个没有工作流运行信息`, + 'run-not-fetched': (count: number) => `${count} 个来自未加载的运行`, + 'not-in-run': (count: number) => + `${count} 个在其运行中没有 gpu_metrics 产物或 power-audit 数据包(产物已过期,或使用其他采集器)`, + } satisfies Record string>, + missingUndrawn: '未绘制', + noTraces: '当前可见硬件没有遥测曲线。请在图例中启用一个系列或选择其他日期。', + noArtifacts: '这些数据点早于按配置上传的遥测产物,因此没有可用的时间线。', + droppedRuns: (runs: number) => + `另有 ${runs} 个运行的遥测数据未加载(每张图表最多 ${POWER_TIMELINE_MAX_RUNS} 个运行)。`, + telemetry: '遥测来源', + method: + '整个基准测试任务期间每个 GPU 板卡功耗(nvidia-smi / amd-smi,Slurm / Dynamo 运行为 DCGM)的一秒平均值;加粗段为实测平均值所依据的有效测量窗口。虚线:硬件注册表中各硬件的额定 TDP。', + methodPools: + '在 GPU 池模式下,每条线是一个 worker 角色池(预填充或解码 GPU)的功耗总和,虚线参考为池内 GPU 数量 × 额定 TDP。', + instructions: 'Shift+滚轮横向缩放 · 拖动平移 · 双击重置 · 点击数据点固定提示框', + dismiss: '点击其他区域关闭', + phase: { + before: '测量窗口之前(启动 / warmup)', + window: '测量窗口内', + after: '测量窗口之后', + unknown: '未记录测量窗口', + } satisfies Record, + meanPerGpu: '每 GPU 平均', + gpus: (count: number) => `${count} 个 GPU`, + min: '最小', + max: '最大', + validated: '有效平均值', + sinceStart: '距起点', + tdp: 'TDP', + allIn: 'all-in', + poolShort: { prefill: '预填充', decode: '解码', all: '全部 GPU' } satisfies Record< + PowerPoolRole, + string + >, + yPool: 'GPU 池功耗(W)', + pool: 'GPU 池', + poolPower: '池功耗', + poolTdp: '池 TDP', + focused: (label: string) => `聚焦:${label}`, + showAll: '显示全部', + unofficialRun: '非官方运行', + branch: '分支', + viewWorkflow: '查看工作流运行', + }, +} as const; + +type XMode = 'wall' | 'elapsed'; +type LineMode = 'mean' | 'gpu' | 'pool'; + +interface TimelineSample { + trace: PowerTimelineTrace; + color: string; + overlayIndex: number | null; + column: number; + timeMs: number; + /** Data-space x for the active mode: epoch ms (wall) or seconds (elapsed). */ + x: number; + /** Mean watts across the GPUs sampled in the bucket; the pool's summed watts in pool mode. */ + y: number; + min: number; + max: number; + /** GPUs with a sample in the bucket (inside the pool, in pool mode). */ + gpuCount: number; + phase: WindowPhase; + /** The pool this sample sums and its device count, in pool mode. */ + pool?: { role: PowerPoolRole; gpuCount: number }; +} + +interface TracePoint { + x: number; + y: number | null; +} + +interface TracePath { + id: string; + traceKey: string; + hwKey: string; + overlayIndex: number | null; + color: string; + segment: 'full' | 'window'; + width: number; + opacity: number; + points: TracePoint[]; + /** Worker-role pool the line sums, in pool mode. */ + pool?: PowerPoolRole; +} + +interface ReferenceLine { + id: string; + watts: number; + label: string; + color: string; + kind: 'tdp' | 'utility'; + /** Pools the line is sized for, in pool mode (roles sharing one GPU count). */ + pools?: PowerPoolRole[]; +} + +/** Vertical distance between stacked reference labels that share a watts value. */ +const REFERENCE_LABEL_ROW = 13; + +interface TraceLabel { + /** Join key: the trace key, plus the pool role in pool mode. */ + id: string; + traceKey: string; + hwKey: string; + pool?: PowerPoolRole; + color: string; + text: string; + x: number; + y: number; +} + +interface DrawModel { + paths: TracePath[]; + labels: TraceLabel[]; +} + +export interface PowerTimelineProps { + chartId: string; + /** Official points of the chart (display-limit clipped points restored). */ + data: InferenceData[]; + overlayData?: OverlayData; + yLabel: string; + caption?: React.ReactNode; +} + +async function fetchPowerSeries( + request: PowerTimelineRequest, + signal: AbortSignal, +): Promise { + const params = new URLSearchParams({ runId: request.runId, series: 'power' }); + if (request.prefix) params.set('prefix', request.prefix); + const response = await fetch(`/api/gpu-metrics?${params.toString()}`, { + cache: 'no-store', + signal, + }); + const body = (await response.json()) as GpuPowerSeriesResponse | { error: string }; + if (!response.ok) { + throw new Error('error' in body ? body.error : `HTTP ${response.status}`); + } + return body as GpuPowerSeriesResponse; +} + +function formatElapsed(totalSeconds: number): string { + const seconds = Math.max(0, Math.round(totalSeconds)); + const h = Math.floor(seconds / 3600); + const m = Math.floor((seconds % 3600) / 60); + const s = seconds % 60; + const mm = h > 0 ? String(m).padStart(2, '0') : String(m); + return `${h > 0 ? `${h}:` : ''}${mm}:${String(s).padStart(2, '0')}`; +} + +const formatUtcClock = d3.utcFormat('%H:%M:%S'); +const formatUtcDate = d3.utcFormat('%Y-%m-%d'); +/** Pool sums run to thousands of watts; group the digits. */ +const formatWatts = d3.format(',.0f'); + +function baseHardware(hwKey: string): string { + return hwKey.split('_')[0]; +} + +/** + * The pools a trace draws in pool mode: its worker-role pools, or every GPU as + * one pool when the collector assigned no roles (a single-node trace then shows + * its deployment total on the same axis). + */ +function drawnPools(series: GpuPowerSeries): PowerPool[] { + const pools = tracePools(series); + return pools.length > 0 ? pools : [allGpuPool(series)]; +} + +/** SVG dash of a pool line: per role from the comparison palette; `all` stays solid. */ +function poolDash(pool: PowerPoolRole): string | null { + const dash = powerVariantDash({ kind: 'role', id: pool }); + return dash === '' ? null : dash; +} + +interface TraceRow { + id: string; + pool?: PowerPoolRole; + values: (number | null)[]; +} + +/** One polyline's values per line mode: the GPU mean, each GPU, or each pool's sum. */ +function traceRows(series: GpuPowerSeries, lineMode: LineMode): TraceRow[] { + if (lineMode === 'gpu') { + return series.gpus.map((gpu, row) => ({ id: `gpu${gpu}`, values: series.power[row] })); + } + if (lineMode === 'pool') { + return drawnPools(series).map((pool) => ({ + id: `pool:${pool.role}`, + pool: pool.role, + values: series.t.map((_, column) => sumPowerAt(series, pool.rows, column)), + })); + } + return [{ id: 'mean', values: series.t.map((_, column) => meanPowerAt(series, column)) }]; +} + +/** Builds the mean, per-GPU or per-pool polylines plus the window emphasis for one trace. */ +function tracePaths( + trace: PowerTimelineTrace, + color: string, + overlayIndex: number | null, + xMode: XMode, + lineMode: LineMode, +): TracePath[] { + const { series } = trace; + const xOf = (column: number) => + xMode === 'wall' ? bucketTimeMs(series, column) : series.t[column] - series.t[0]; + const rows = traceRows(series, lineMode); + const faint = lineMode === 'gpu' ? 0.22 : 0.32; + const strong = lineMode === 'gpu' ? 0.85 : 1; + const widths = lineMode === 'gpu' ? [1, 1.5] : [1.25, 2.25]; + const paths: TracePath[] = []; + for (const row of rows) { + const full: TracePoint[] = []; + const window: TracePoint[] = []; + row.values.forEach((value, column) => { + const point = { x: xOf(column), y: value }; + full.push(point); + if (windowPhase(trace, bucketTimeMs(series, column)) === 'window') window.push(point); + }); + paths.push({ + id: `${trace.key}:${row.id}:full`, + traceKey: trace.key, + hwKey: trace.point.hwKey, + overlayIndex, + color, + segment: 'full', + width: widths[0], + opacity: faint, + points: full, + pool: row.pool, + }); + if (window.length > 1) { + paths.push({ + id: `${trace.key}:${row.id}:window`, + traceKey: trace.key, + hwKey: trace.point.hwKey, + overlayIndex, + color, + segment: 'window', + width: widths[1], + opacity: strong, + points: window, + pool: row.pool, + }); + } + } + return paths; +} + +/** + * Hover targets of one trace: one stream over all its GPUs (mean watts), or in + * pool mode one stream per pool (summed watts) so the tooltip can name the pool. + */ +function traceSamples( + trace: PowerTimelineTrace, + color: string, + overlayIndex: number | null, + xMode: XMode, + lineMode: LineMode, +): TimelineSample[] { + const { series } = trace; + if (lineMode !== 'pool') { + const rows = series.power.map((_, row) => row); + return sampleRows(trace, color, overlayIndex, xMode, rows, undefined); + } + return drawnPools(series).flatMap((pool) => + sampleRows(trace, color, overlayIndex, xMode, pool.rows, { + role: pool.role, + gpuCount: pool.rows.length, + }), + ); +} + +function sampleRows( + trace: PowerTimelineTrace, + color: string, + overlayIndex: number | null, + xMode: XMode, + rows: readonly number[], + pool: TimelineSample['pool'], +): TimelineSample[] { + const { series } = trace; + const samples: TimelineSample[] = []; + for (let column = 0; column < series.t.length; column++) { + let sum = 0; + let count = 0; + let min = Number.POSITIVE_INFINITY; + let max = Number.NEGATIVE_INFINITY; + for (const row of rows) { + const value = series.power[row]?.[column]; + if (value === null || value === undefined) continue; + sum += value; + count += 1; + if (value < min) min = value; + if (value > max) max = value; + } + // A pool bucket missing a device is a gap, as in `sumPowerAt`, not a dip. + if (count === 0 || (pool && count < rows.length)) continue; + const timeMs = bucketTimeMs(series, column); + samples.push({ + trace, + color, + overlayIndex, + column, + timeMs, + x: xMode === 'wall' ? timeMs : series.t[column] - series.t[0], + y: pool ? sum : sum / count, + min, + max, + gpuCount: count, + phase: windowPhase(trace, timeMs), + pool, + }); + } + return lttbDownsample( + samples, + HIT_POINTS_PER_TRACE, + (sample) => sample.x, + (sample) => sample.y, + ); +} + +type AnyContinuousScale = d3.ScaleLinear; + +function drawTraces( + group: d3.Selection, + xScale: AnyContinuousScale, + yScale: AnyContinuousScale, + model: DrawModel, + highlight: string | null, +): void { + const line = d3 + .line() + .defined((point) => point.y !== null) + .x((point) => xScale(point.x)) + .y((point) => yScale(point.y ?? 0)) + .curve(d3.curveLinear); + const selection = group + .selectAll('path.power-trace') + .data(model.paths, (path) => path.id); + selection.exit().remove(); + selection + .enter() + .append('path') + .attr('class', 'power-trace') + .attr('fill', 'none') + .attr('stroke-linejoin', 'round') + .attr('stroke-linecap', 'round') + .merge(selection) + .attr('data-trace-key', (path) => path.traceKey) + .attr('data-hw', (path) => path.hwKey) + .attr('data-segment', (path) => path.segment) + .attr('data-run-index', (path) => (path.overlayIndex === null ? null : path.overlayIndex)) + .attr('data-pool', (path) => path.pool ?? null) + .attr('stroke-dasharray', (path) => (path.pool ? poolDash(path.pool) : null)) + .attr('stroke', (path) => path.color) + .attr('stroke-width', (path) => path.width) + .attr('opacity', (path) => traceOpacity(path, highlight)) + .attr('d', (path) => line(path.points)); +} + +function traceOpacity(path: TracePath, highlight: string | null): number { + if (highlight === null) return path.opacity; + return highlight === path.hwKey || highlight === path.traceKey + ? Math.min(1, path.opacity + 0.15) + : path.opacity * 0.15; +} + +function drawLabels( + group: d3.Selection, + xScale: AnyContinuousScale, + yScale: AnyContinuousScale, + model: DrawModel, + plotWidth: number, + highlight: string | null, +): void { + const selection = group + .selectAll('text.power-trace-label') + .data(model.labels, (label) => label.id); + selection.exit().remove(); + selection + .enter() + .append('text') + .attr('class', 'power-trace-label') + .attr('font-family', CHART_FONT_SANS) + .attr('font-size', px(CHART_TYPE.dataLabel)) + .attr('font-weight', '600') + .attr('dominant-baseline', 'middle') + .attr('pointer-events', 'none') + .merge(selection) + .attr('data-hw', (label) => label.hwKey) + .attr('data-pool', (label) => label.pool ?? null) + .attr('fill', (label) => label.color) + .attr('opacity', (label) => + highlight === null || highlight === label.hwKey || highlight === label.traceKey ? 1 : 0.2, + ) + .text((label) => label.text) + .each(function (label) { + const x = xScale(label.x); + const y = yScale(label.y); + // Sit just past the last sample; flip inside the plot near the right edge. + const overflow = x + 6 + label.text.length * 6.5 > plotWidth; + d3.select(this) + .attr('text-anchor', overflow ? 'end' : 'start') + .attr('x', overflow ? x - 6 : x + 6) + .attr('y', y); + }); +} + +function drawReferenceLines( + group: d3.Selection, + yScale: AnyContinuousScale, + width: number, + lines: ReferenceLine[], +): void { + group.selectAll('.power-reference').remove(); + const slots = referenceLabelSlots(lines); + lines.forEach((line, index) => { + const y = yScale(line.watts); + const g = group + .append('g') + .attr('class', 'power-reference') + .attr('data-reference', line.kind) + .attr('data-pool', line.pools?.join(' ') ?? null) + .attr('data-watts', line.watts); + g.append('line') + .attr('x1', 0) + .attr('x2', width) + .attr('y1', y) + .attr('y2', y) + .attr('stroke', line.color) + .attr('stroke-width', 1.25) + .attr('stroke-dasharray', line.kind === 'tdp' ? '6,4' : '2,4') + .attr('opacity', 0.9); + g.append('text') + .attr('x', width - 4) + .attr('y', y - 5 - slots[index] * REFERENCE_LABEL_ROW) + .attr('text-anchor', 'end') + .attr('fill', line.color) + .attr('font-family', CHART_FONT_SANS) + .attr('font-size', px(CHART_TYPE.annotation)) + .attr('font-weight', '600') + .text(line.label); + }); +} + +/** Reasons, then the undrawn configs (named when few, counted per hardware when many). */ +function describeMissing( + missing: readonly MissingTrace[], + t: (typeof STRINGS)[keyof typeof STRINGS], + hardwareLabel: (point: InferenceData) => string, +): string { + const reasons = new Map(); + for (const { reason } of missing) reasons.set(reason, (reasons.get(reason) ?? 0) + 1); + const reasonText = [...reasons.entries()] + .map(([reason, count]) => t.missingReason[reason](count)) + .join('; '); + let list: string; + if (missing.length <= MAX_LISTED_MISSING) { + list = missing + .map(({ point }) => `${hardwareLabel(point)} ${traceConfigLabel(point)}`) + .join(' · '); + } else { + const perHardware = new Map(); + for (const { point } of missing) { + const label = hardwareLabel(point); + perHardware.set(label, (perHardware.get(label) ?? 0) + 1); + } + list = [...perHardware.entries()].map(([label, count]) => `${label} ×${count}`).join(' · '); + } + return `${reasonText}. ${t.missingUndrawn}: ${list}`; +} + +export default function PowerTimeline({ + chartId, + data, + overlayData, + yLabel, + caption, +}: PowerTimelineProps) { + const locale = useLocale(); + const t = STRINGS[locale]; + const { hardwareConfig, hwTypesWithData } = useInferenceData(); + const { activeHwTypes, selectedPrecisions, quickFilters } = useInferenceFilters(); + const { isLegendExpanded, highContrast } = useInferenceDisplay(); + const { setBestPerSku, toggleHwType, setIsLegendExpanded } = useInferenceActions(); + const { + unofficialRunInfos, + runIndexByUrl, + activeOverlayHwTypes, + localOfficialOverride, + setUnifiedOverlaySelection, + } = useUnofficialRun(); + + const [xModeChoice, setXModeChoice] = useState(null); + const [lineMode, setLineMode] = useState('mean'); + const [showUtility, setShowUtility] = useState(false); + const [highlight, setHighlight] = useState(null); + /** Trace a "View power trace" deep link asked for, once the join has produced it. */ + const [focusKey, setFocusKey] = useState(null); + /** The deep-link request, read once on mount; `undefined` until read, `null` once honoured. */ + const requestedFocusRef = useRef(undefined); + + // The chart's point list still carries every precision, quick-filtered rows + // and rows without a validated average (ScatterGraph applies those gates at + // draw time); only rows that plot on the measured-average axis for the + // current selection have a trace to look up. + const plotsHere = useCallback( + (point: InferenceData) => + point.measuredPowerTimeline !== undefined && + selectedPrecisions.includes(point.precision) && + matchesQuickFilters(point, quickFilters), + [selectedPrecisions, quickFilters], + ); + const measuredData = useMemo(() => data.filter(plotsHere), [data, plotsHere]); + const overlayPoints = useMemo( + () => (overlayData?.data ?? []).filter(plotsHere), + [overlayData, plotsHere], + ); + const overlayPointSet = useMemo(() => new Set(overlayPoints), [overlayPoints]); + const allPoints = useMemo( + () => [...measuredData, ...overlayPoints], + [measuredData, overlayPoints], + ); + + const hwKeysInData = useMemo( + () => + [...new Set(measuredData.map((point) => point.hwKey))].toSorted( + (a, b) => getModelSortIndex(a) - getModelSortIndex(b) || a.localeCompare(b), + ), + [measuredData], + ); + const stableHcKeys = useMemo(() => [...hwTypesWithData], [hwTypesWithData]); + const activeOfficialKeys = useMemo(() => [...activeHwTypes], [activeHwTypes]); + const { resolveColor, getCssColor } = useThemeColors({ + highContrast, + identifiers: hwKeysInData, + activeKeys: activeOfficialKeys, + hcKeys: stableHcKeys, + }); + + // ── Telemetry fetch: one request per workflow run ────────────────────────── + const requests = useMemo(() => planPowerTimelineRequests(allPoints), [allPoints]); + // The deep-link request is read once, before planning, so its run is fetched + // even when the chart spans more runs than the cap. + if (requestedFocusRef.current === undefined) { + requestedFocusRef.current = consumePowerTraceFocus(); + } + const focusRunRef = useRef(traceKeyRunId(requestedFocusRef.current)); + // Overlay runs were requested explicitly (`?unofficialrun=`), so they take + // the cap's slots before official runs; the deep-linked run still goes first. + const overlayRunIds = useMemo( + () => + new Set( + overlayPoints + .map((point) => runIdFromUrl(point.run_url)) + .filter((runId): runId is string => runId !== null), + ), + [overlayPoints], + ); + const fetchedRequests = useMemo( + () => + prioritizeRun(prioritizeRuns(requests, overlayRunIds), focusRunRef.current).slice( + 0, + POWER_TIMELINE_MAX_RUNS, + ), + [requests, overlayRunIds], + ); + const droppedRuns = requests.length - fetchedRequests.length; + const queries = useQueries({ + queries: fetchedRequests.map((request) => ({ + queryKey: ['power-timeline', request.runId, request.prefix] as const, + queryFn: ({ signal }: { signal: AbortSignal }) => fetchPowerSeries(request, signal), + staleTime: 5 * 60_000, + retry: 1, + })), + }); + const loadingRuns = queries.filter((query) => query.isPending).length; + const errors = fetchedRequests + .map((request, index) => ({ request, error: queries[index].error })) + .filter( + (entry): entry is { request: PowerTimelineRequest; error: Error } => + entry.error instanceof Error, + ); + const responses = useMemo(() => { + const map = new Map(); + fetchedRequests.forEach((request, index) => { + const response = queries[index].data; + if (response) map.set(request.runId, response); + }); + return map; + // Query data objects are stable per fetch; deriving from them keeps the map memoised. + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [fetchedRequests, ...queries.map((query) => query.data)]); + + const { traces, missing } = useMemo( + () => joinPowerTimeline(allPoints, responses), + [allPoints, responses], + ); + const hasAnyArtifact = requests.length > 0; + + // Honour the deep link once its trace exists; a disaggregated trace opens in + // pool mode because its prefill / decode split is what the reader came for. + useEffect(() => { + const requested = requestedFocusRef.current; + if (!requested) return; + const trace = traces.find((entry) => entry.key === requested); + if (!trace) return; + requestedFocusRef.current = null; + setFocusKey(trace.key); + if (tracePools(trace.series).length > 0) setLineMode('pool'); + }, [traces]); + + useEffect(() => { + if (loadingRuns > 0 || responses.size === 0) return; + track('inference_power_timeline_loaded', { + traces: traces.length, + missing: missing.length, + runs: responses.size, + }); + }, [loadingRuns, responses.size, traces.length, missing.length]); + + // ── Visible traces and their colours ──────────────────────────────────────── + const colorForTrace = useCallback( + (trace: PowerTimelineTrace): { color: string; overlayIndex: number | null } => { + if (overlayPointSet.has(trace.point)) { + const index = overlayRunIndex(trace.point.run_url ?? null, runIndexByUrl); + return { color: overlayRunColor(index), overlayIndex: index }; + } + return { color: getCssColor(resolveColor(trace.point.hwKey)), overlayIndex: null }; + }, + [overlayPointSet, runIndexByUrl, getCssColor, resolveColor], + ); + // Same visibility source as ScatterGraph: an overlay session may hold a + // local official selection that has not been written back to the filters. + const officialHwTypes = localOfficialOverride ?? activeHwTypes; + // With an overlay loaded the chart reads localOfficialOverride, so a legend + // click must write the unified selection the way ScatterGraph does; the + // context's toggleHwType would change activeHwTypes with no visible effect. + const handleToggleHwType = useCallback( + (key: string) => { + if (!overlayData) { + toggleHwType(key); + return; + } + setBestPerSku(false, { applySelection: false }); + const official = new Set([...officialHwTypes].filter((hw) => hwTypesWithData.has(hw))); + setUnifiedOverlaySelection( + computeToggle(official, key, hwTypesWithData), + activeOverlayHwTypes, + ); + }, + [ + overlayData, + toggleHwType, + setBestPerSku, + officialHwTypes, + hwTypesWithData, + setUnifiedOverlaySelection, + activeOverlayHwTypes, + ], + ); + const visibleTraces = useMemo( + () => + traces.filter((trace) => + overlayPointSet.has(trace.point) + ? activeOverlayHwTypes.has(trace.point.hwKey) + : officialHwTypes.has(trace.point.hwKey), + ), + [traces, overlayPointSet, activeOverlayHwTypes, officialHwTypes], + ); + // Focus follows visibility: hiding the focused hardware in the legend lifts + // the dimming and the chip instead of dimming everything with nothing lit. + const focusedTrace = useMemo( + () => visibleTraces.find((trace) => trace.key === focusKey) ?? null, + [visibleTraces, focusKey], + ); + /** Legend hover wins over the deep-link focus while it lasts. */ + const activeHighlight = highlight ?? focusedTrace?.key ?? null; + const visibleRunCount = useMemo( + () => new Set(visibleTraces.map((trace) => trace.runId)).size, + [visibleTraces], + ); + const xMode: XMode = xModeChoice ?? (visibleRunCount <= 1 ? 'wall' : 'elapsed'); + + // The pools switch is offered only where a visible trace carries worker roles; + // pool mode left without one would draw deployment totals with no way back. + const hasPools = useMemo( + () => visibleTraces.some((trace) => tracePools(trace.series).length > 0), + [visibleTraces], + ); + useEffect(() => { + if (lineMode === 'pool' && !hasPools && visibleTraces.length > 0) setLineMode('mean'); + }, [lineMode, hasPools, visibleTraces.length]); + + const model = useMemo(() => { + const paths: TracePath[] = []; + const labels: TraceLabel[] = []; + for (const trace of visibleTraces) { + const { color, overlayIndex } = colorForTrace(trace); + const tracePathSet = tracePaths(trace, color, overlayIndex, xMode, lineMode); + paths.push(...tracePathSet); + if (visibleTraces.length > MAX_LABELED_TRACES) continue; + // One end label per trace; per pool in pool mode, so the role reads off the line. + const groups: { pool?: PowerPoolRole; paths: TracePath[] }[] = + lineMode === 'pool' + ? drawnPools(trace.series).map((pool) => ({ + pool: pool.role, + paths: tracePathSet.filter((path) => path.pool === pool.role), + })) + : [{ paths: tracePathSet }]; + for (const group of groups) { + const anchor = + group.paths.find((path) => path.segment === 'window') ?? + group.paths.find((path) => path.segment === 'full'); + const last = anchor?.points.filter((point) => point.y !== null).at(-1); + if (!last || last.y === null) continue; + labels.push({ + id: group.pool ? `${trace.key}:${group.pool}` : trace.key, + traceKey: trace.key, + hwKey: trace.point.hwKey, + pool: group.pool, + color, + text: group.pool + ? `c${trace.point.conc} · ${t.poolShort[group.pool]}` + : `c${trace.point.conc}`, + x: last.x, + y: last.y, + }); + } + } + return { paths, labels }; + }, [visibleTraces, colorForTrace, xMode, lineMode, t]); + + const samples = useMemo( + () => + visibleTraces.flatMap((trace) => { + const { color, overlayIndex } = colorForTrace(trace); + return traceSamples(trace, color, overlayIndex, xMode, lineMode); + }), + [visibleTraces, colorForTrace, xMode, lineMode], + ); + + // Rated references: per hardware in mean / per-GPU modes; per (hardware, + // pool size) in pool mode, scaled to the pool so the summed line and its + // ceiling share the axis. Roles of one hardware that hold the same number + // of GPUs share a ceiling and draw as one line (`prefill / decode ×16`). + const referenceLines = useMemo(() => { + const lines: ReferenceLine[] = []; + const pushLines = (base: string, color: string, pool?: PoolSizeGroup) => { + const specs = HW_REGISTRY[base]; + if (!specs) return; + const label = specs.label ?? base.toUpperCase(); + const size = pool?.size ?? 1; + const id = pool ? `${base}:${pool.roles.join('+')}:${pool.size}` : base; + const name = pool + ? `${label} ${pool.roles.map((role) => t.poolShort[role]).join(' / ')} ×${pool.size}` + : label; + if (specs.tdp > 0) { + lines.push({ + id: `tdp:${id}`, + watts: specs.tdp * size, + label: `${name} ${t.tdp} ${specs.tdp * size} W`, + color, + kind: 'tdp', + pools: pool?.roles, + }); + } + if (showUtility && specs.power > 0) { + const watts = pool ? Math.round(specs.power * 1000) * size : specs.power * 1000; + lines.push({ + id: `utility:${id}`, + watts, + label: `${name} ${t.allIn} ${Math.round(watts)} W`, + color, + kind: 'utility', + pools: pool?.roles, + }); + } + }; + // First trace of a hardware sets the reference colour, as before. + const perBase = new Map(); + for (const trace of visibleTraces) { + const base = baseHardware(trace.point.hwKey); + if (!perBase.has(base)) perBase.set(base, { color: colorForTrace(trace).color, pools: [] }); + if (lineMode === 'pool') perBase.get(base)!.pools.push(...drawnPools(trace.series)); + } + for (const [base, { color, pools }] of perBase) { + if (lineMode !== 'pool') { + pushLines(base, color); + continue; + } + for (const group of groupPoolsBySize(pools)) pushLines(base, color, group); + } + return lines; + }, [visibleTraces, colorForTrace, showUtility, lineMode, t]); + + // ── Scales ───────────────────────────────────────────────────────────────── + const xDomain = useMemo<[number, number]>(() => { + let min = Number.POSITIVE_INFINITY; + let max = Number.NEGATIVE_INFINITY; + for (const path of model.paths) { + if (path.segment !== 'full') continue; + for (const point of path.points) { + if (point.x < min) min = point.x; + if (point.x > max) max = point.x; + } + } + if (!Number.isFinite(min) || !Number.isFinite(max)) { + return xMode === 'wall' ? [Date.UTC(2026, 0, 1), Date.UTC(2026, 0, 1, 0, 10)] : [0, 600]; + } + return min === max ? [min, max + (xMode === 'wall' ? 60_000 : 60)] : [min, max]; + }, [model.paths, xMode]); + const yDomain = useMemo<[number, number]>(() => { + let max = 0; + for (const path of model.paths) { + for (const point of path.points) if (point.y !== null && point.y > max) max = point.y; + } + for (const line of referenceLines) if (line.watts > max) max = line.watts; + return [0, max > 0 ? max * 1.06 : 100]; + }, [model.paths, referenceLines]); + + const xTickFormat = useMemo(() => { + if (xMode === 'elapsed') return (value: d3.AxisDomain) => formatElapsed(Number(value)); + const span = xDomain[1] - xDomain[0]; + const format = d3.utcFormat(span < 3 * 60_000 ? '%H:%M:%S' : '%H:%M'); + return (value: d3.AxisDomain) => + format(value instanceof Date ? value : new Date(Number(value))); + }, [xMode, xDomain]); + + // ── Layers ───────────────────────────────────────────────────────────────── + const highlightRef = useRef(activeHighlight); + highlightRef.current = activeHighlight; + const layers = useMemo[]>( + () => [ + { + type: 'custom', + key: 'power-reference-lines', + render: (group, ctx) => { + drawReferenceLines(group, ctx.yScale as AnyContinuousScale, ctx.width, referenceLines); + }, + }, + { + type: 'custom', + key: 'power-traces', + render: (group, ctx: RenderContext) => { + drawTraces( + group, + ctx.xScale as AnyContinuousScale, + ctx.yScale as AnyContinuousScale, + model, + highlightRef.current, + ); + }, + onZoom: (group, ctx: ZoomContext) => { + drawTraces( + group, + ctx.newXScale as AnyContinuousScale, + ctx.newYScale as AnyContinuousScale, + model, + highlightRef.current, + ); + }, + }, + { + type: 'point', + key: 'power-hit-points', + data: samples, + config: { + getCx: () => 0, + getCy: () => 0, + getX: (sample) => sample.x, + getY: (sample) => sample.y, + getColor: (sample) => sample.color, + getRadius: () => 2, + // Pool streams of one trace share columns; the role keeps their keys apart. + keyFn: (sample) => + sample.pool + ? `${sample.trace.key}:${sample.pool.role}:${sample.column}` + : `${sample.trace.key}:${sample.column}`, + maxPoints: Number.POSITIVE_INFINITY, + }, + }, + { + type: 'custom', + key: 'power-trace-labels', + render: (group, ctx: RenderContext) => { + drawLabels( + group, + ctx.xScale as AnyContinuousScale, + ctx.yScale as AnyContinuousScale, + model, + ctx.width, + highlightRef.current, + ); + }, + onZoom: (group, ctx: ZoomContext) => { + drawLabels( + group, + ctx.newXScale as AnyContinuousScale, + ctx.newYScale as AnyContinuousScale, + model, + ctx.width, + highlightRef.current, + ); + }, + }, + ], + [referenceLines, model, samples], + ); + + const onDisplayUpdate = useCallback( + (ctx: RenderContext) => { + const root = d3.select(ctx.layout.svg.node() as SVGSVGElement); + root + .selectAll('path.power-trace') + .attr('opacity', (path) => traceOpacity(path, activeHighlight)); + root + .selectAll('text.power-trace-label') + .attr('opacity', (label) => + activeHighlight === null || + activeHighlight === label.hwKey || + activeHighlight === label.traceKey + ? 1 + : 0.2, + ); + }, + [activeHighlight], + ); + + const hardwareLabel = useCallback( + (point: InferenceData): string => { + const config = overlayPointSet.has(point) + ? overlayData?.hardwareConfig[point.hwKey] + : hardwareConfig[point.hwKey]; + return config ? getDisplayLabel(config) : point.hwKey; + }, + [overlayPointSet, overlayData, hardwareConfig], + ); + + const tooltipContent = useCallback( + (sample: TimelineSample, isPinned: boolean) => { + const { trace } = sample; + const point = trace.point; + const overlayInfo = + sample.overlayIndex === null ? null : unofficialRunInfos[sample.overlayIndex]; + const tdp = HW_REGISTRY[baseHardware(point.hwKey)]?.tdp ?? 0; + const elapsed = formatElapsed((sample.timeMs - trace.series.startMs) / 1000); + const clock = `${formatUtcClock(new Date(sample.timeMs))} UTC`; + const time = + xMode === 'wall' ? `${clock} · +${elapsed} ${t.sinceStart}` : `+${elapsed} · ${clock}`; + const colon = locale === 'zh' ? ':' : ':'; + const validated = point.measuredAvgPower?.y; + const { pool } = sample; + const readings = pool + ? `
    ${t.pool}${colon} ${t.poolShort[pool.role]} · ${t.gpus(pool.gpuCount)}
    +
    ${t.poolPower}${colon} ${formatWatts(sample.y)} W${ + tdp > 0 + ? ` (${((sample.y / (tdp * pool.gpuCount)) * 100).toFixed(0)}% ${t.poolTdp})` + : '' + }
    +
    ${t.meanPerGpu}${colon} ${(sample.y / sample.gpuCount).toFixed(1)} W · ${t.min} ${sample.min.toFixed(1)} W · ${t.max} ${sample.max.toFixed(1)} W
    ` + : `
    ${t.meanPerGpu}${colon} ${sample.y.toFixed(1)} W${ + tdp > 0 + ? ` (${((sample.y / tdp) * 100).toFixed(0)}% ${t.tdp})` + : '' + }
    +
    ${t.gpus(sample.gpuCount)} · ${t.min} ${sample.min.toFixed(1)} W · ${t.max} ${sample.max.toFixed(1)} W
    `; + return `
    + ${isPinned ? `
    ${t.dismiss}
    ` : ''} +
    ${hardwareLabel(point)} · ${traceConfigLabel(point)}${ + overlayInfo ? ` · ✕ ${overlayInfo.branch || `run ${overlayInfo.id}`}` : '' + }
    +
    ${time}
    + ${readings} +
    ${t.phase[sample.phase]}
    + ${ + typeof validated === 'number' + ? `
    ${t.validated}${colon} ${validated.toFixed(1)} W
    ` + : '' + } +
    `; + }, + [unofficialRunInfos, xMode, t, locale, hardwareLabel], + ); + + // ── Legend ───────────────────────────────────────────────────────────────── + const legendItems = useMemo(() => { + const overlayItems = + overlayData && unofficialRunInfos.length > 0 + ? unofficialRunInfos + .map((info, index) => { + const hasPoints = overlayPoints.some( + (point) => overlayRunIndex(point.run_url ?? null, runIndexByUrl) === index, + ); + if (!hasPoints) return null; + const branch = info.branch || `run ${info.id}`; + return { + name: `✕ unofficial-run-${info.id}`, + label: `✕ ${branch}`, + color: overlayRunColor(index), + title: `${t.unofficialRun}: ${branch}`, + isHighlighted: true, + hw: `overlay-run-${info.id}`, + isActive: true, + isRemovable: false, + onClick: () => {}, + tooltip: ( +
    +
    {t.unofficialRun}
    +
    + {t.branch}: {branch} +
    + {info.url && ( + + {t.viewWorkflow} + + )} +
    + ), + }; + }) + .filter((item): item is NonNullable => item !== null) + : []; + const officialItems = hwKeysInData + .filter((key) => hwTypesWithData.has(key) && hardwareConfig[key]) + .map((key) => { + const config = hardwareConfig[key]; + return { + name: config.name, + label: getDisplayLabel(config), + color: resolveColor(key), + title: config.gpu, + hw: key, + isActive: officialHwTypes.has(key), + onClick: () => { + handleToggleHwType(key); + track('latency_hw_type_toggled', { hw: key }); + }, + tooltip: null, + }; + }); + return [...overlayItems, ...officialItems]; + }, [ + overlayData, + unofficialRunInfos, + overlayPoints, + runIndexByUrl, + hwKeysInData, + hwTypesWithData, + hardwareConfig, + resolveColor, + officialHwTypes, + handleToggleHwType, + t, + ]); + + // Per-GPU and pools are two views of the same lines, so either switch turns + // the other off; both fall back to the mean. + const chooseLineMode = (next: LineMode) => { + setLineMode(next); + track('inference_power_timeline_lines_changed', { lines: next }); + }; + const switches: LegendSwitchConfig[] = [ + { + id: 'power-timeline-per-gpu', + label: t.perGpu, + checked: lineMode === 'gpu', + onCheckedChange: (checked) => chooseLineMode(checked ? 'gpu' : 'mean'), + infoTooltip: t.perGpuHelp, + }, + ]; + if (hasPools) { + switches.push({ + id: 'power-timeline-pools', + label: t.pools, + checked: lineMode === 'pool', + onCheckedChange: (checked) => chooseLineMode(checked ? 'pool' : 'mean'), + infoTooltip: t.poolsHelp, + }); + } + switches.push({ + id: 'power-timeline-utility', + label: t.utilityLines, + checked: showUtility, + onCheckedChange: (checked) => { + setShowUtility(checked); + track('inference_power_timeline_utility_toggled', { enabled: checked }); + }, + infoTooltip: t.utilityHelp, + }); + + const legendElement = ( + { + setIsLegendExpanded(expanded); + track('latency_legend_expanded', { expanded }); + }} + onItemHover={(id) => setHighlight(id)} + onItemHoverEnd={() => setHighlight(null)} + hideAtomFootnote + switches={switches} + /> + ); + + const runInfos = useMemo( + () => + fetchedRequests + .map((request) => responses.get(request.runId)?.runInfo) + .filter((info): info is NonNullable => Boolean(info)), + [fetchedRequests, responses], + ); + + const toolbar = ( +
    + {t.timeAxis} + + value={xMode} + ariaLabel={t.timeAxis} + role="group" + options={[ + { value: 'wall', label: t.wall, testId: 'power-timeline-axis-wall' }, + { value: 'elapsed', label: t.elapsed, testId: 'power-timeline-axis-elapsed' }, + ]} + onValueChange={(mode) => { + setXModeChoice(mode); + track('inference_power_timeline_axis_changed', { mode }); + }} + /> +
    + ); + + let emptyMessage: string | null = null; + if (loadingRuns > 0) emptyMessage = t.loading(loadingRuns); + else if (!hasAnyArtifact) emptyMessage = t.noArtifacts; + else if (visibleTraces.length === 0) emptyMessage = t.noTraces; + + return ( +
    + + key={`${chartId}-${xMode}-${lineMode}`} + chartId={chartId} + data={samples} + height={CHART_HEIGHT} + margin={MARGIN} + watermark={overlayData && overlayPoints.length > 0 ? 'unofficial' : 'logo'} + testId="power-timeline-chart-svg" + grabCursor + instructions={t.instructions} + xScale={ + xMode === 'wall' + ? { type: 'time', domain: [new Date(xDomain[0]), new Date(xDomain[1])] } + : { type: 'linear', domain: xDomain } + } + yScale={{ type: 'linear', domain: yDomain, nice: true }} + xAxis={{ + label: xMode === 'wall' ? t.xWall : t.xElapsed, + tickCount: 10, + tickFormat: xTickFormat, + }} + yAxis={{ label: lineMode === 'pool' ? t.yPool : yLabel, tickCount: 8 }} + layers={layers} + displayIdentity={activeHighlight ?? ''} + onDisplayUpdate={onDisplayUpdate} + zoom={{ + enabled: true, + axes: 'x', + scaleExtent: [1, 60], + resetEventName: `power_timeline_zoom_reset_${chartId}`, + }} + tooltip={{ + rulerType: 'crosshair', + content: tooltipContent, + getRulerX: (sample, xScale) => (xScale as AnyContinuousScale)(sample.x), + getRulerY: (sample, yScale) => yScale(sample.y), + onHoverStart: (selection) => { + selection.attr('r', 5).attr('stroke', 'white').attr('stroke-width', 1); + }, + onHoverEnd: (selection) => { + selection.attr('r', 2).attr('stroke', 'none'); + }, + attachToLayer: 2, + }} + legendElement={legendElement} + caption={ + <> + {caption} + {toolbar} + + } + noDataOverlay={ + emptyMessage ? ( +
    + {emptyMessage} +
    + ) : undefined + } + /> +
    + {focusedTrace && ( +

    + + {t.focused( + `${hardwareLabel(focusedTrace.point)} · ${traceConfigLabel(focusedTrace.point)}`, + )} + + +

    + )} + {errors.map(({ request, error }) => ( +

    + {t.loadError(request.runId, error.message)} +

    + ))} + {droppedRuns > 0 &&

    {t.droppedRuns(droppedRuns)}

    } + {loadingRuns === 0 && missing.length > 0 && traces.length > 0 && ( +

    + {t.missing(missing.length, traces.length + missing.length)}{' '} + {describeMissing(missing, t, hardwareLabel)} +

    + )} + {runInfos.length > 0 && ( +

    + {t.telemetry}:{' '} + {runInfos.map((info, index) => ( + + {index > 0 && ' · '} + + {`run ${info.id}`} + + {info.createdAt ? ` (${formatUtcDate(new Date(info.createdAt))})` : ''} + + ))} +

    + )} +

    {t.method}

    + {hasPools &&

    {t.methodPools}

    } +
    +
    + ); +} diff --git a/packages/app/src/components/inference/ui/ScatterGraph.tsx b/packages/app/src/components/inference/ui/ScatterGraph.tsx index 4f67e2fbe..04206c80f 100644 --- a/packages/app/src/components/inference/ui/ScatterGraph.tsx +++ b/packages/app/src/components/inference/ui/ScatterGraph.tsx @@ -16,6 +16,7 @@ import { useInferenceDisplay, useInferenceFilters, } from '@/components/inference/InferenceContext'; +import { usePerfRulerStore } from '@/components/inference/perf-ruler-store'; import { useTraceAvailability } from '@/hooks/api/use-trace-availability'; import { useLogAvailability } from '@/hooks/api/use-log-availability'; import { computeToggle } from '@/hooks/useTogglableSet'; @@ -47,7 +48,13 @@ import { matchKnownConfigIssues, pointMatchesIssue } from '@/lib/known-issues'; import { useLocale } from '@/lib/use-locale'; import { getLineLabelVendorIcon } from '@/lib/vendor-logos'; import { formatNumber, getDisplayLabel, updateRepoUrl } from '@/lib/utils'; -import { getInferenceHardwareConfig, getInferenceRunLabel } from '@/lib/inference-labels'; +import { + getInferenceHardwareConfig, + getInferenceRunLabel, + getOverlayLineLabel, + OVERLAY_LABEL_MARKER, + overlayRunTag, +} from '@/lib/inference-labels'; import { D3Chart } from '@/lib/d3-chart/D3Chart'; import type { CustomLayerConfig, @@ -75,6 +82,7 @@ import { renderPerfRulers, type PerfRulerEndInput, type PerfRulerGeometry, + type PerfRulerMeasurement, type PerfRulerRenderEntry, type PerfRulerState, } from '@/lib/d3-chart/layers/perf-ruler'; @@ -109,6 +117,7 @@ import { chartFrontier, upperPowerEnvelope, isPowerCurveMetric, + isPowerGaugeSeries, isMeasuredPowerCurveMetric, } from '@/components/inference/utils/powerCurves'; import type { @@ -116,17 +125,38 @@ import type { ClippedInferenceData, InferenceData, ScatterGraphProps, + PowerVariant, } from '@/components/inference/types'; import { generateOverlayTooltipContent, generateTooltipContent, } from '@/components/inference/utils/tooltipUtils'; +import { + POWER_TIMELINE_METRIC_KEY, + requestPowerTraceFocus, + traceKeyForPoint, +} from '@/components/inference/utils/powerTimeline'; import { QuickFiltersDialog } from '@/components/inference/ui/QuickFiltersDialog'; import { ScatterEmptyState } from '@/components/inference/ui/ScatterEmptyState'; import { scatterPointConfigId, scatterPointJoinId, + parseScatterSeriesKey, + scatterSeriesKey, } from '@/components/inference/utils/point-identity'; +import { + flatSeriesValue, + inferPowerCompare, + lineLabelHardwareKey, + lineLabelSeriesId, + metricPlotsWatts, + powerCompareBase, + powerLineLabelSuffix, + powerVariantDash, + powerVariantId, + powerVariantLabel, + powerVariantsInData, +} from '@/components/inference/utils/power-compare'; import LegendPointsDialog from '@/components/inference/ui/LegendPointsDialog'; import { renderOffloadHalo } from '@/components/inference/utils/offload-halo'; import { renderLegacyPowerRing } from '@/components/inference/utils/legacy-power-marker'; @@ -219,6 +249,32 @@ const optimalPointKey = (d: InferenceData): string => const EMPTY_OVERLAY_DATA: InferenceData[] = []; const EMPTY_CLIPPED_DATA: ClippedInferenceData[] = []; +/** + * Legend ids of the comparison-series rows (`i_pcompare`), distinct from + * hardware keys so the shared hover / toggle handlers can tell them apart. + */ +const POWER_VARIANT_LEGEND_PREFIX = 'power-variant:'; +/** Comparison clones sit behind the base series they annotate. */ +const POWER_VARIANT_POINT_OPACITY = 0.6; +const pointOpacityForVariant = (d: InferenceData): number => + d.powerVariant ? POWER_VARIANT_POINT_OPACITY : 1; +/** Dash for a series key's variant id (`parseScatterSeriesKey().variant`). */ +const powerVariantDashById = (variantId: string | null | undefined): string => + variantId ? (VARIANT_DASH_BY_ID.get(variantId) ?? '') : ''; +const VARIANT_DASH_BY_ID = new Map( + ( + [ + ['basis', 'gpu-measured'], + ['basis', 'gpu-provisioned'], + ['basis', 'utility-provisioned'], + ['basis', 'utility-modeled'], + ['role', 'all'], + ['role', 'prefill'], + ['role', 'decode'], + ] as const + ).map(([kind, id]) => [id, powerVariantDash({ kind, id } as PowerVariant)]), +); + function setsEqual(a: Set, b: Set): boolean { if (a.size !== b.size) return false; for (const value of a) { @@ -531,6 +587,7 @@ const ScatterGraph = React.memo( setQuickFilterDeployment, setQuickFilterSpec, setQuickFilterPower, + setSelectedYAxisMetric, } = useInferenceActions(); const paretoDirection = chartDefinition[`${selectedYAxisMetric}_roofline`] as | ParetoDirection @@ -549,15 +606,28 @@ const ScatterGraph = React.memo( const groups = groupPointsByDate(points); if (showPowerEnvelope) { for (const [date, samples] of groups) { - groups.set(date, upperPowerEnvelope(samples, chartDefinition.chartType !== 'e2e')); + groups.set( + date, + upperPowerEnvelope( + samples, + chartDefinition.chartType !== 'e2e', + isPowerGaugeSeries(selectedYAxisMetric, samples[0]), + ), + ); } } return groups; }, - [showPowerEnvelope, chartDefinition.chartType], + [showPowerEnvelope, chartDefinition.chartType, selectedYAxisMetric], ); const locale = useLocale(); const legendT = SCATTER_STRINGS[locale]; + // Comparison series (`i_pcompare`) switched off from the legend. Chart-local, + // like Optimal Only's point set: the URL carries the comparison, not which + // of its rows a reader hid while looking. + const [hiddenPowerVariants, setHiddenPowerVariants] = useState>( + () => new Set(), + ); const ephemeralUrlState = useEphemeralUrlState(); const costLimit = chartDefinition.y_cost_limit ?? 0; const latencyLimit = chartDefinition.y_latency_limit ?? 0; @@ -798,7 +868,7 @@ const ScatterGraph = React.memo( () => data.reduce( (acc, point) => { - const key = `${point.hwKey}_${point.precision}`; + const key = scatterSeriesKey(point); if (!acc[key]) acc[key] = []; acc[key].push(point); return acc; @@ -935,7 +1005,7 @@ const ScatterGraph = React.memo( } const buckets = new Map(); const getBucket = (point: InferenceData) => { - const key = `${point.hwKey}|${point.precision}|${point.date}`; + const key = `${scatterSeriesKey(point)}|${point.date}`; let bucket = buckets.get(key); if (!bucket) { bucket = { @@ -984,7 +1054,7 @@ const ScatterGraph = React.memo( const buckets = new Map(); const getBucket = (point: InferenceData) => { const runIndex = overlayRunIndex(point.run_url ?? null, runIndexByUrl); - const key = `${point.hwKey}|${point.precision}|${point.date}|run${runIndex}`; + const key = `${scatterSeriesKey(point)}|${point.date}|run${runIndex}`; let bucket = buckets.get(key); if (!bucket) { bucket = { @@ -1089,6 +1159,8 @@ const ScatterGraph = React.memo( interface Entry { hwKey: string; runIndex: number; + /** Comparison variant id for boundary / role clones, null for the run's base series. */ + variant: string | null; points: InferenceData[]; } if (processedOverlayData.length === 0) return {} as Record; @@ -1097,8 +1169,15 @@ const ScatterGraph = React.memo( const grouped = processedOverlayData.reduce( (acc, p) => { const runIndex = overlayRunIndex(p.run_url ?? null, runIndexByUrl); - const key = `${p.hwKey}_${p.precision}_run${runIndex}`; - if (!acc[key]) acc[key] = { hwKey: String(p.hwKey), runIndex, points: [] }; + const key = `${scatterSeriesKey(p)}_run${runIndex}`; + if (!acc[key]) { + acc[key] = { + hwKey: String(p.hwKey), + runIndex, + variant: p.powerVariant?.id ?? null, + points: [], + }; + } acc[key].points.push(p); return acc; }, @@ -1177,9 +1256,11 @@ const ScatterGraph = React.memo( // its X marker sitting on the dashed roofline and read as a pareto point. const isOverlayPointVisible = useCallback( (d: InferenceData) => + !hiddenPowerVariants.has(powerVariantId(d.powerVariant)) && (!hideNonOptimal || overlayOptimalPoints.has(d)) && (!showPowerEnvelope || showAllMeasurements || overlayEnvelopePoints.has(d)), [ + hiddenPowerVariants, hideNonOptimal, overlayOptimalPoints, showPowerEnvelope, @@ -1214,6 +1295,44 @@ const ScatterGraph = React.memo( const { data: persistedLogAvailability } = useLogAvailability(persistedPointIds); const [fixedLogPointId, setFixedLogPointId] = useState(null); + // "View power trace" on a pinned tooltip (official or overlay point): the + // same-tab click stays in-page — remember which trace to emphasise, switch + // the metric to the Timeline display, and let the anchor's href keep + // serving open-in-new-tab. Listeners are attached per pin because the + // tooltip HTML is replaced on every pin. + const attachPowerTraceAction = useCallback( + (tooltipEl: HTMLElement, d: InferenceData, overlay: boolean) => { + const action = tooltipEl.querySelector('[data-action="view-power-trace"]'); + const traceKey = traceKeyForPoint(d); + if (!action || !traceKey) return; + action.addEventListener('click', (actionEvent) => { + actionEvent.stopPropagation(); + // Modifier / auxiliary clicks keep the anchor's own behaviour: the + // href opens this chart's timeline in a new tab or window. + const mouse = actionEvent as MouseEvent; + if ( + mouse.button !== 0 || + mouse.metaKey || + mouse.ctrlKey || + mouse.shiftKey || + mouse.altKey + ) { + return; + } + actionEvent.preventDefault(); + requestPowerTraceFocus(traceKey); + chartRef.current?.dismissTooltip(); + setSelectedYAxisMetric(POWER_TIMELINE_METRIC_KEY); + track('inference_power_trace_opened', { + hwKey: String(d.hwKey), + conc: d.conc, + overlay, + }); + }); + }, + [setSelectedYAxisMetric], + ); + // --- Legend points table (per-series drill-down opened from the legend) --- const [pointsTableTarget, setPointsTableTarget] = useState(null); const [quickFiltersOpen, setQuickFiltersOpen] = useState(false); @@ -1248,6 +1367,7 @@ const ScatterGraph = React.memo( const pts = pointsData.filter( (p) => p.hwKey === hwKey && + !p.powerVariant && selectedPrecisions.includes(p.precision) && (!hideNonOptimal || optimalPointKeys.has(optimalPointKey(p))), ); @@ -1263,6 +1383,7 @@ const ScatterGraph = React.memo( const pts = processedOverlayData.filter( (p) => overlayRunIndex(p.run_url ?? null, runIndexByUrl) === runIndex && + !p.powerVariant && activeOverlayHwTypes.has(p.hwKey as string) && (!hideNonOptimal || overlayOptimalPoints.has(p)), ); @@ -1483,6 +1604,7 @@ const ScatterGraph = React.memo( (d: InferenceData) => effectiveActiveHwTypes.has(d.hwKey as string) && selectedPrecisions.includes(d.precision) && + !hiddenPowerVariants.has(powerVariantId(d.powerVariant)) && (!hideNonOptimal || optimalPointKeys.has(optimalPointKey(d))) && (!showPowerEnvelope || showAllMeasurements || @@ -1490,6 +1612,7 @@ const ScatterGraph = React.memo( [ effectiveActiveHwTypes, selectedPrecisions, + hiddenPowerVariants, hideNonOptimal, optimalPointKeys, showPowerEnvelope, @@ -1672,12 +1795,69 @@ const ScatterGraph = React.memo( getCssColor, ]); + // The comparison in effect and the base series' identity under it. The + // base is the selected metric's own series; deriving it from which variant + // no official point carries breaks when only an overlay carries the + // comparison, and line labels need the same answer as the legend rows. + const powerCompareMode = useMemo(() => { + const official = inferPowerCompare(pointsData); + return official === 'none' ? inferPowerCompare(processedOverlayData) : official; + }, [pointsData, processedOverlayData]); + const powerCompareBaseId = useMemo( + () => powerVariantId(powerCompareBase(selectedYAxisMetric, powerCompareMode)), + [selectedYAxisMetric, powerCompareMode], + ); + + // One legend row per comparison series present (base first). Rows toggle + // chart-local visibility and hover-highlight that series across hardware. + const powerVariantLegendItems = useMemo(() => { + const allPoints = [...pointsData, ...processedOverlayData]; + const variants = powerVariantsInData(allPoints, selectedYAxisMetric); + const baseId = powerCompareBaseId; + return variants.map((variant) => { + const id = powerVariantId(variant); + const legendId = `${POWER_VARIANT_LEGEND_PREFIX}${id}`; + const isBase = id === baseId; + return { + name: legendId, + hw: legendId, + label: powerVariantLabel(variant, locale), + color: 'var(--foreground)', + // The base series is solid, like its points; siblings carry their dash. + lineDasharray: isBase ? '1 0' : powerVariantDash(variant) || '1 0', + isActive: !hiddenPowerVariants.has(isBase ? '' : id), + isRemovable: false, + onClick: () => { + const key = isBase ? '' : id; + setHiddenPowerVariants((prev) => { + const next = new Set(prev); + if (next.has(key)) next.delete(key); + else next.add(key); + return next; + }); + track('inference_power_compare_series_toggled', { + series: id, + visible: hiddenPowerVariants.has(key), + }); + }, + }; + }); + }, [ + pointsData, + processedOverlayData, + selectedYAxisMetric, + powerCompareBaseId, + locale, + hiddenPowerVariants, + ]); + const powerTierCounts = useMemo(() => { - const officialTotal = pointsData.filter((point) => - selectedPrecisions.includes(point.precision), + // Comparison clones re-plot the same measurements; count each once. + const officialTotal = pointsData.filter( + (point) => !point.powerVariant && selectedPrecisions.includes(point.precision), ); - const overlayTotal = processedOverlayData.filter((point) => - selectedPrecisions.includes(point.precision), + const overlayTotal = processedOverlayData.filter( + (point) => !point.powerVariant && selectedPrecisions.includes(point.precision), ); const officialVisible = officialTotal.filter(isPointVisible); const overlayVisible = overlayTotal.filter( @@ -1702,9 +1882,13 @@ const ScatterGraph = React.memo( const hw = el.dataset.hwKey; const prec = el.dataset.precision; if (hw === null || hw === undefined || prec === null || prec === undefined) return false; - return effectiveActiveHwTypes.has(hw) && selectedPrecisions.includes(prec); + return ( + effectiveActiveHwTypes.has(hw) && + selectedPrecisions.includes(prec) && + !hiddenPowerVariants.has(el.dataset.powerVariant ?? '') + ); }, - [effectiveActiveHwTypes, selectedPrecisions], + [effectiveActiveHwTypes, selectedPrecisions, hiddenPowerVariants], ); // --- Interaction state ref --- @@ -1721,6 +1905,7 @@ const ScatterGraph = React.memo( isPointVisible, isOverlayPointVisible, effectiveActiveHwTypes, + hiddenPowerVariants, selectedPrecisions, activeOverlayHwTypes, getCssColor, @@ -1733,6 +1918,7 @@ const ScatterGraph = React.memo( isPointVisible, isOverlayPointVisible, effectiveActiveHwTypes, + hiddenPowerVariants, selectedPrecisions, activeOverlayHwTypes, getCssColor, @@ -1751,17 +1937,39 @@ const ScatterGraph = React.memo( // the curves' rendered paths at the iso-x — neither end needs to be a // data point. Multiple rulers accumulate (capped in the pure module); // completing one immediately allows starting the next. - const [preferPerfRulerMode, setPerfRulerMode] = useState(false); + // + // The primary chart's rulers live in the InferenceProvider store so they + // ride along in share links (`i_rulers`) and survive a remount (table + // view toggle). Every other instance — the replay chart, which draws the + // same curve classes, and harnesses mounted without the provider — keeps + // component-local state. Both paths share one `[state, setState]` pair + // below, so the reducers, refs, and draw passes are path-agnostic. + const perfRulerStore = usePerfRulerStore(); + const persistedRulers = perfRulerStore?.chartId === chartId ? perfRulerStore : undefined; + // Rulers only render while the mode is on (and the mode-off effect below + // clears them), so restored share-link rulers — pending or already + // committed by a previous mount — switch the mode on for this instance. + const [preferPerfRulerMode, setPerfRulerMode] = useState( + () => + persistedRulers !== undefined && + (persistedRulers.pending !== null || persistedRulers.state.rulers.length > 0), + ); const perfRulerMode = preferPerfRulerMode && (!showPowerEnvelope || isMeasuredPowerAxis); - const [perfRulerState, setPerfRulerState] = useState(EMPTY_PERF_RULER_STATE); + const [localPerfRulerState, setLocalPerfRulerState] = + useState(EMPTY_PERF_RULER_STATE); + const perfRulerState = persistedRulers ? persistedRulers.state : localPerfRulerState; + const setPerfRulerState = persistedRulers ? persistedRulers.setState : setLocalPerfRulerState; // Changing the x- or y-axis metric (including the x percentile, which // `x_scale_field` encodes) clears every ruler: the curves are redrawn // in different units, so a ruler that persisted would measure a ratio // the user never placed. Runs before the draw pass so no stale ruler - // ever paints over the new curves. + // ever paints over the new curves. Render-time adjustment is only legal + // for this component's own state, so the hook targets the local state; + // the store applies the same reset to persisted rulers inside the + // provider (see usePerfRulerStoreValue). usePerfRulerAxisReset( perfRulerAxisMetricKey(chartDefinition.x_scale_field, selectedYAxisMetric), - setPerfRulerState, + setLocalPerfRulerState, ); // Draw passes read mode/state through refs so toggling off clears the // rulers in the same pre-paint layout pass — lines/labels must never @@ -1827,6 +2035,51 @@ const ScatterGraph = React.memo( [], ); + // Share-link rulers commit only once BOTH curve paths are in the DOM — + // otherwise the prune pass would eat them before their data (i_gpus, + // comparison dates, overlay runs) has arrived. Hidden curves (opacity 0) + // count as present, like for prune. The iso-x is clamped to the pair's + // overlap through the drawn paths, so a rounded or since-shifted iso-x + // still renders; a pair with disjoint spans can never be measured on + // these axes and is dropped. This runs from the draw pass rather than a + // React effect: the chart first draws in a D3Chart-local re-render + // (dimensions are measured after mount), which re-renders nothing here, + // so an effect keyed on our props could miss the first draw and leave + // resolvable rulers pending for the rest of the session. The store is + // read through a ref for the same reason the draw passes read the ruler + // state through refs. Nothing commits while the mode is off (forced off + // by the power envelope, or switched off by the user) — the mode-off + // effect discards pending rulers, and the analytics event must not + // report a restore nobody saw. Draw passes can repeat before React has + // applied a commit, so the pending list handed over is remembered by + // identity and skipped until the store replaces it. + const persistedRulersRef = useRef(persistedRulers); + persistedRulersRef.current = persistedRulers; + const committedPendingRef = useRef(null); + const commitPendingPerfRulers = useCallback( + (zoomGroup: d3.Selection) => { + const store = persistedRulersRef.current; + const pending = store?.pending ?? null; + if (!store || !pending || !perfRulerModeRef.current) return; + if (committedPendingRef.current === pending) return; + const curveExists = (cls: string) => !zoomGroup.select(`.${CSS.escape(cls)}`).empty(); + const resolved: PerfRulerMeasurement[] = []; + const remaining: PerfRulerMeasurement[] = []; + for (const ruler of pending) { + if (!curveExists(ruler.curveA) || !curveExists(ruler.curveB)) { + remaining.push(ruler); + continue; + } + const isoX = clampPerfRulerIsoXToOverlap(ruler.curveA, ruler.curveB, ruler.isoX); + if (isoX !== null) resolved.push({ ...ruler, isoX }); + } + if (remaining.length === pending.length) return; + committedPendingRef.current = pending; + store.commitPending(resolved, remaining.length > 0 ? remaining : null); + }, + [clampPerfRulerIsoXToOverlap], + ); + // Curve click (widened hit strokes): iso-x is the click's x pixel // through the CURRENT rendered x scale, stored in data space. const handlePerfRulerCurveClick = useCallback( @@ -1855,7 +2108,7 @@ const ScatterGraph = React.memo( (point: InferenceData, source: 'official' | 'overlay') => { const ctx = perfRulerDrawCtxRef.current; if (!ctx) return; - const series = `${String(point.hwKey)}_${point.precision}`; + const series = scatterSeriesKey(point); const base = source === 'overlay' ? `overlay-roofline-${series}_run${overlayRunIndex(point.run_url ?? null, runIndexByUrl)}` @@ -1885,8 +2138,12 @@ const ScatterGraph = React.memo( // the switch handler also clears synchronously, this covers // programmatic mode changes). `clearPerfRulers` bails out with the same // reference when there is nothing to clear. + // Share-link rulers still waiting for their curves go too — the user + // switched the tool off, so nothing should surface later. useEffect(() => { - if (!perfRulerMode) setPerfRulerState(clearPerfRulers); + if (perfRulerMode) return; + setPerfRulerState(clearPerfRulers); + persistedRulers?.discardPending(); }, [perfRulerMode]); // Invisible widened hit strokes over every rendered roofline path @@ -2036,6 +2293,7 @@ const ScatterGraph = React.memo( ) => { perfRulerDrawCtxRef.current = { zoomGroup, xScale, yScale, width, height }; syncPerfRulerHitPaths(zoomGroup); + commitPendingPerfRulers(zoomGroup); const state = perfRulerStateRef.current; const entries: PerfRulerRenderEntry[] = []; if (perfRulerModeRef.current && state.rulers.length > 0) { @@ -2089,7 +2347,7 @@ const ScatterGraph = React.memo( ); if (!dragHandles.empty()) dragHandles.call(perfRulerDrag); }, - [syncPerfRulerHitPaths, perfRulerDrag], + [syncPerfRulerHitPaths, commitPendingPerfRulers, perfRulerDrag], ); drawPerfRulerRef.current = drawPerfRuler; @@ -2123,16 +2381,29 @@ const ScatterGraph = React.memo( const svg = chartRef.current?.getSvgElement?.(); if (!svg) return; const root = d3.select(svg); + // A comparison-series legend row highlights that boundary / role + // across every hardware instead of one hardware across series. + const variantId = hwKey.startsWith(POWER_VARIANT_LEGEND_PREFIX) + ? hwKey.slice(POWER_VARIANT_LEGEND_PREFIX.length) + : null; + const matchesPoint = (d: InferenceData) => + variantId === null + ? String(d.hwKey) === hwKey + : powerVariantId(d.powerVariant) === variantId; root .selectAll('.dot-group') .style('opacity', (d) => - isPointVisible(d) ? (String(d.hwKey) === hwKey ? 1 : 0.15) : 0, + isPointVisible(d) ? (matchesPoint(d) ? pointOpacityForVariant(d) : 0.15) : 0, ); root .selectAll('.roofline-path, .official-overflow-continuation') .style('opacity', function () { if (!isRooflineVisible(this)) return 0; - return this.dataset.hwKey === hwKey ? null : '0.15'; + const matches = + variantId === null + ? this.dataset.hwKey === hwKey + : (this.dataset.powerVariant ?? '') === variantId; + return matches ? null : '0.15'; }); root .selectAll('.parallelism-label, .line-label') @@ -2149,7 +2420,7 @@ const ScatterGraph = React.memo( const root = d3.select(svg); root .selectAll('.dot-group') - .style('opacity', (d) => (isPointVisible(d) ? 1 : 0)); + .style('opacity', (d) => (isPointVisible(d) ? pointOpacityForVariant(d) : 0)); root .selectAll('.roofline-path, .official-overflow-continuation') .style('opacity', function () { @@ -2162,9 +2433,16 @@ const ScatterGraph = React.memo( (this as SVGGElement).dataset, effectiveActiveHwTypes, selectedPrecisions, + activeOverlayHwTypes, ); }); - }, [isPointVisible, isRooflineVisible, effectiveActiveHwTypes, selectedPrecisions]); + }, [ + isPointVisible, + isRooflineVisible, + effectiveActiveHwTypes, + selectedPrecisions, + activeOverlayHwTypes, + ]); // --- Zoom config --- const eventPrefix = chartDefinition.chartType === 'e2e' ? 'latency' : 'interactivity'; @@ -2311,10 +2589,12 @@ const ScatterGraph = React.memo( }); }); } + attachPowerTraceAction(tooltipEl, d, false); }, attachToLayer: 1, // scatter layer is index 1 (after rooflines at 0) }), [ + attachPowerTraceAction, xLabel, yLabel, selectedYAxisMetric, @@ -2328,6 +2608,31 @@ const ScatterGraph = React.memo( // --- Layers --- const layers = useMemo((): LayerConfig[] => { + // Line-label identity of one drawn series under a power comparison + // (`i_pcompare`): the base series keeps the hardware key, so pinned + // anchors and hover hooks keep working; a sibling is `::`. + const wattsAxis = metricPlotsWatts(selectedYAxisMetric); + const lineLabelIdentity = (hw: string, points: readonly InferenceData[]) => { + const variant = points[0]?.powerVariant; + const variantId = powerVariantId(variant); + const isBase = !variant || variantId === powerCompareBaseId; + return { variant, variantId, isBase, seriesId: lineLabelSeriesId(hw, variant, isBase) }; + }; + // A sibling's label says which series it is; a flat provisioned boundary + // (TDP, all-in) on a watts axis also states its value. + const lineLabelSuffix = ( + identity: ReturnType, + points: readonly InferenceData[], + ) => + powerLineLabelSuffix(identity.variant, { + isBase: identity.isBase, + locale, + flatWatts: + !identity.isBase && wattsAxis && identity.variant?.kind === 'basis' + ? flatSeriesValue(points.map((point) => point.y)) + : null, + }); + // ── Layer 0: Rooflines + gradient labels (custom) ── const rooflineLayer: CustomLayerConfig = { type: 'custom', @@ -2365,6 +2670,8 @@ const ScatterGraph = React.memo( key: string; hw: string; precision: string; + /** Comparison variant id (`i_pcompare`), '' for the base series. */ + variant: string; points: InferenceData[]; stroke: string; visible: boolean; @@ -2373,10 +2680,11 @@ const ScatterGraph = React.memo( const activeGradientIds = new Set(); Object.entries(displayedRooflines).forEach(([key, pts]) => { - const hw = key.split('_').slice(0, -1).join('_'); - const precision = key.split('_').pop()!; + const { hw, precision, variant } = parseScatterSeriesKey(key); const visible = - ir.effectiveActiveHwTypes.has(hw) && ir.selectedPrecisions.includes(precision); + ir.effectiveActiveHwTypes.has(hw) && + ir.selectedPrecisions.includes(precision) && + !ir.hiddenPowerVariants.has(variant ?? ''); const baseStroke = ir.getCssColor(ir.resolveColor(hw)); // Split into per-date sub-paths so the line never crosses dates. @@ -2421,6 +2729,7 @@ const ScatterGraph = React.memo( key: entryKey, hw, precision, + variant: variant ?? '', points: datePoints, stroke, visible, @@ -2448,9 +2757,12 @@ const ScatterGraph = React.memo( .attr('data-curve-kind', showPowerEnvelope ? 'power-envelope' : 'pareto') .attr('data-hw-key', (d) => d.hw) .attr('data-precision', (d) => d.precision) + .attr('data-power-variant', (d) => d.variant || null) .attr('fill', 'none') .attr('stroke', (d) => d.stroke) .attr('stroke-width', 2.5) + // Comparison siblings share the hardware colour; the dash tells them apart. + .attr('stroke-dasharray', (d) => powerVariantDashById(d.variant) || null) .attr('d', (d) => lineGen(d.points)) .style('transition', 'opacity 150ms ease') .style('opacity', (d) => (d.visible ? 1 : 0)); @@ -2471,10 +2783,11 @@ const ScatterGraph = React.memo( if (showGradientLabels) { Object.entries(allPointLabelsByKey).forEach(([key, pointLabels]) => { if (pointLabels.length < 2) return; - const hw = key.split('_').slice(0, -1).join('_'); - const precision = key.split('_').pop()!; + const { hw, precision, variant } = parseScatterSeriesKey(key); const visible = - ir.effectiveActiveHwTypes.has(hw) && ir.selectedPrecisions.includes(precision); + ir.effectiveActiveHwTypes.has(hw) && + ir.selectedPrecisions.includes(precision) && + !ir.hiddenPowerVariants.has(variant ?? ''); const segments: { label: string; color: string; points: InferenceData[] }[] = []; let cur = { @@ -2572,12 +2885,22 @@ const ScatterGraph = React.memo( // ── Line labels (run name along each roofline) ── let lineLabels: LineLabelPlacement[] = []; + // Comparison variant and label suffix per label key, for the text + // segments and the `data-power-variant` hook on each pill. + const lineLabelMeta = new Map< + string, + { variantId: string; suffix: string; runTag: string } + >(); if (showLineLabels) { const multiPrecision = ir.selectedPrecisions.length > 1; const officialByGroup = new Map(); for (const entry of entries) { if (!entry.visible) continue; - const groupKey = multiPrecision ? entry.key : entry.hw; + // One label per hardware and, under a power comparison, per + // sibling series: the measured line and its boundary / pool + // lines each say which one they are, instead of the longest + // line taking the hardware's only label. + const groupKey = multiPrecision ? entry.key : `${entry.hw}::${entry.variant}`; const previous = officialByGroup.get(groupKey); if (!previous || entry.points.length > previous.points.length) { officialByGroup.set(groupKey, entry); @@ -2586,39 +2909,59 @@ const ScatterGraph = React.memo( const officialSeries: LineLabelSeries[] = [ ...officialByGroup.values(), - ].map((entry) => ({ - key: entry.key, - seriesId: entry.hw, - label: lineLabelText( - entry.hw, - entry.precision, - multiPrecision, - modelLabel, - entry.points, - ), - color: ir.getCssColor(ir.resolveColor(entry.hw)), - points: entry.points, - keepVisibleOnCollision: entry.points.length === 1, - })); + ].map((entry) => { + const identity = lineLabelIdentity(entry.hw, entry.points); + const suffix = lineLabelSuffix(identity, entry.points); + lineLabelMeta.set(entry.key, { variantId: identity.variantId, suffix, runTag: '' }); + return { + key: entry.key, + seriesId: identity.seriesId, + label: `${lineLabelText( + entry.hw, + entry.precision, + multiPrecision, + modelLabel, + entry.points, + )}${suffix}`, + color: ir.getCssColor(ir.resolveColor(entry.hw)), + points: entry.points, + keepVisibleOnCollision: entry.points.length === 1, + }; + }); + // Runs drawing the same hardware need a run tag on their pills. + const overlayRunsByHw = new Map>(); + for (const group of Object.values(displayedOverlayRooflines)) { + if (!ir.activeOverlayHwTypes.has(group.hwKey)) continue; + if (!overlayRunsByHw.has(group.hwKey)) overlayRunsByHw.set(group.hwKey, new Set()); + overlayRunsByHw.get(group.hwKey)!.add(group.runIndex); + } const overlaySeries: LineLabelSeries[] = Object.entries( displayedOverlayRooflines, ).flatMap(([overlayKey, group]) => { if (!ir.activeOverlayHwTypes.has(group.hwKey)) return []; const info = unofficialRunInfos[group.runIndex]; const precision = group.points[0]?.precision ?? ''; - const runLabel = info - ? getInferenceRunLabel(`✕ ${info.branch || `run ${info.id}`}`, group.points) - : ''; + const hardwareLabel = lineLabelText( + group.hwKey, + precision, + multiPrecision, + modelLabel, + group.points, + ); + const sharesHardware = (overlayRunsByHw.get(group.hwKey)?.size ?? 0) > 1; + const runTag = info && sharesHardware ? overlayRunTag(info) : ''; const label = info - ? multiPrecision - ? `${runLabel} ${getPrecisionLabel(precision as Precision)}` - : runLabel - : lineLabelText(group.hwKey, precision, multiPrecision, modelLabel, group.points); + ? getOverlayLineLabel(hardwareLabel, info, sharesHardware) + : hardwareLabel; + const identity = lineLabelIdentity(group.hwKey, group.points); + const suffix = lineLabelSuffix(identity, group.points); + const key = `overlay-${overlayKey}`; + lineLabelMeta.set(key, { variantId: identity.variantId, suffix, runTag }); return [ { - key: `overlay-${overlayKey}`, - seriesId: group.hwKey, - label, + key, + seriesId: identity.seriesId, + label: `${label}${suffix}`, color: overlayRunColor(group.runIndex), points: group.points, }, @@ -2641,16 +2984,19 @@ const ScatterGraph = React.memo( const labeledKeys = new Set(lineLabels.map((label) => label.key)); for (const entry of entries) { if (labeledKeys.has(entry.key)) continue; + const identity = lineLabelIdentity(entry.hw, entry.points); + const suffix = lineLabelSuffix(identity, entry.points); + lineLabelMeta.set(entry.key, { variantId: identity.variantId, suffix, runTag: '' }); lineLabels.push({ key: entry.key, - seriesId: entry.hw, - label: lineLabelText( + seriesId: identity.seriesId, + label: `${lineLabelText( entry.hw, entry.precision, multiPrecision, modelLabel, entry.points, - ), + )}${suffix}`, color: ir.getCssColor(ir.resolveColor(entry.hw)), x: xScale(entry.points[0].x), y: yScale(entry.points[0].y), @@ -2667,20 +3013,35 @@ const ScatterGraph = React.memo( } renderLineLabels(zoomGroup, lineLabels, { - seriesAttribute: 'data-hw-key', - iconFor: (label) => getLineLabelVendorIcon(label.seriesId), + seriesAttribute: 'data-series-id', + iconFor: (label) => getLineLabelVendorIcon(lineLabelHardwareKey(label.seriesId)), configureGroup: (labelGroup, label) => { labelGroup .attr('data-visible', label.visible ? '1' : '0') + // Legend hover and filter sync key labels by hardware alone; + // the variant names the comparison sibling ('' for the base). + .attr('data-hw-key', lineLabelHardwareKey(label.seriesId)) + .attr('data-power-variant', lineLabelMeta.get(label.key)?.variantId ?? '') .select('.ll-bg') .attr('opacity', 0.95); }, configureText: (text, label) => { - const config = getHardwareConfig(label.seriesId, modelLabel); + const config = getHardwareConfig(lineLabelHardwareKey(label.seriesId), modelLabel); + // Parse the hardware part without the variant suffix, which gets + // its own segment so the engine is still matched at the end. + const meta = lineLabelMeta.get(label.key); + const suffix = meta?.suffix ?? ''; + const runTag = meta?.runTag ?? ''; + let coreLabel = suffix ? label.label.slice(0, -suffix.length) : label.label; + if (runTag) coreLabel = coreLabel.slice(0, -runTag.length); + // Overlay pills lead with the run marker; the hardware behind it is + // parsed like an official pill so the GPU name stays bold. + const marker = coreLabel.startsWith(OVERLAY_LABEL_MARKER) ? OVERLAY_LABEL_MARKER : ''; + coreLabel = coreLabel.slice(marker.length); const hardwareLabel = getDisplayLabel(config); const isHardwareLabel = - label.label === hardwareLabel || label.label.startsWith(`${config.label} `); - const remainingLabel = isHardwareLabel ? label.label.slice(config.label.length) : ''; + coreLabel === hardwareLabel || coreLabel.startsWith(`${config.label} `); + const remainingLabel = isHardwareLabel ? coreLabel.slice(config.label.length) : ''; // Use this curve's resolved suffix, not the generic hwKey label: // official and overlay curves can share a key but differ by run. const engineLabel = @@ -2690,8 +3051,18 @@ const ScatterGraph = React.memo( engineLabel && remainingLabel.endsWith(engineLabel) ? remainingLabel.slice(0, -engineLabel.length) : remainingLabel; + const markerSegments = marker + ? [{ className: 'll-marker', text: marker, fill: 'white', weight: '600' }] + : []; + const runSegments = runTag + ? [{ className: 'll-run', text: runTag, fill: '#d1d5db', weight: '400' }] + : []; + const variantSegments = suffix + ? [{ className: 'll-variant', text: suffix, fill: 'white', weight: '500' }] + : []; const segments = isHardwareLabel ? [ + ...markerSegments, { className: 'll-gpu', text: config.label, fill: 'white', weight: '700' }, ...(precisionLabel ? [ @@ -2713,14 +3084,19 @@ const ScatterGraph = React.memo( }, ] : []), + ...runSegments, + ...variantSegments, ] : [ + ...markerSegments, { className: 'll-plain', - text: label.label, + text: coreLabel, fill: 'white', weight: '600', }, + ...runSegments, + ...variantSegments, ]; text .selectAll('tspan') @@ -2829,11 +3205,11 @@ const ScatterGraph = React.memo( { key: string; seriesId: string; points: InferenceData[] } >(); for (const [key, points] of Object.entries(displayedRooflines)) { - const hardware = key.split('_').slice(0, -1).join('_'); - const precision = key.split('_').pop()!; + const { hw: hardware, precision, variant } = parseScatterSeriesKey(key); if ( !ir.effectiveActiveHwTypes.has(hardware) || - !ir.selectedPrecisions.includes(precision) + !ir.selectedPrecisions.includes(precision) || + ir.hiddenPowerVariants.has(variant ?? '') ) { continue; } @@ -2841,12 +3217,12 @@ const ScatterGraph = React.memo( const singleDate = pointsByDate.size === 1; for (const [date, datePoints] of pointsByDate) { const entryKey = singleDate ? key : `${key}__${encodeURIComponent(date)}`; - const groupKey = multiPrecision ? entryKey : hardware; + const groupKey = multiPrecision ? entryKey : `${hardware}::${variant ?? ''}`; const previous = bestByGroup.get(groupKey); if (!previous || datePoints.length > previous.points.length) { bestByGroup.set(groupKey, { key: entryKey, - seriesId: hardware, + seriesId: lineLabelIdentity(hardware, datePoints).seriesId, points: datePoints, }); } @@ -2867,7 +3243,7 @@ const ScatterGraph = React.memo( ? [ { key: `overlay-${overlayKey}`, - seriesId: group.hwKey, + seriesId: lineLabelIdentity(group.hwKey, group.points).seriesId, label: '', color: '', points: group.points, @@ -2901,7 +3277,8 @@ const ScatterGraph = React.memo( interactionRef.current.getCssColor( interactionRef.current.resolveColor(d.hwKey as string), ), - getOpacity: (d) => (interactionRef.current.isPointVisible(d) ? 1 : 0), + getOpacity: (d) => + interactionRef.current.isPointVisible(d) ? pointOpacityForVariant(d) : 0, getPointerEvents: (d) => (interactionRef.current.isPointVisible(d) ? 'auto' : 'none'), hideLabels: !showPointLabels || showGradientLabels, // Concurrency (C=) is appended only when the advanced @@ -2911,6 +3288,7 @@ const ScatterGraph = React.memo( dataAttrs: { 'hw-key': (d) => String(d.hwKey), precision: (d) => d.precision, + 'power-variant': (d) => d.powerVariant?.id ?? '', // Lets the agentic coach mark pick an anchor out of the DOM // without knowing anything about React state. 'benchmark-type': (d) => d.benchmark_type ?? '', @@ -2989,6 +3367,7 @@ const ScatterGraph = React.memo( points: InferenceData[]; stroke: string; runIndex: number; + variant: string | null; } const ovEntries: OvEntry[] = []; Object.entries(displayedOverlayRooflines).forEach(([key, group]) => { @@ -3000,6 +3379,7 @@ const ScatterGraph = React.memo( // Color by run — same palette entry the legend uses, so they match. stroke: overlayRunColor(group.runIndex), runIndex: group.runIndex, + variant: group.variant, }); } }); @@ -3017,9 +3397,21 @@ const ScatterGraph = React.memo( .attr('fill', 'none') .attr('stroke', (d) => d.stroke) .attr('stroke-width', 2) - .attr('stroke-dasharray', (d) => overlayRooflineDasharray(d.runIndex)) + .attr('data-power-variant', (d) => d.variant) + // The run keeps its colour; a comparison sibling takes the + // variant dash so it reads like its official counterpart. + .attr('stroke-dasharray', (d) => + d.variant + ? powerVariantDashById(d.variant) + : overlayRooflineDasharray(d.runIndex), + ) .attr('d', (d) => lineGen(d.points)) - .style('filter', null); + .style('filter', null) + // Comparison rows hidden from the legend (the decoration effect + // keeps this in step with later toggles). + .style('opacity', (d) => + interactionRef.current.hiddenPowerVariants.has(d.variant ?? '') ? 0 : null, + ); // Overlay X-shape points — index-keyed so every point renders const overlayPoints = zoomGroup @@ -3055,7 +3447,7 @@ const ScatterGraph = React.memo( overlayPoints.each(function (d) { const visible = interactionRef.current.isOverlayPointVisible(d); d3.select(this) - .style('opacity', visible ? 1 : 0) + .style('opacity', visible ? pointOpacityForVariant(d) : 0) .style('pointer-events', visible ? 'auto' : 'none'); }); overlayPoints @@ -3139,6 +3531,9 @@ const ScatterGraph = React.memo( y: point.y, overlay: true, }); + // The shared helper has just rendered the pinned content into + // this element and pinned it via `handle`. + attachPowerTraceAction(ctx.tooltipElement, point, true); }, }); }, @@ -3409,10 +3804,12 @@ const ScatterGraph = React.memo( xLabel, yLabel, selectedYAxisMetric, + powerCompareBaseId, isMeasuredEnergyAxis, chartDefinition, locale, drawPerfRuler, + attachPowerTraceAction, ]); // Layers handle for the decoration effect — lets it re-run individual @@ -3491,7 +3888,9 @@ const ScatterGraph = React.memo( zoomGroup.selectAll('.dot-group').each(function (d) { const point = d3.select(this); const visible = ir.isPointVisible(d); - point.style('opacity', visible ? 1 : 0).style('pointer-events', visible ? 'auto' : 'none'); + point + .style('opacity', visible ? pointOpacityForVariant(d) : 0) + .style('pointer-events', visible ? 'auto' : 'none'); const color = (showGradientLabels && gradientColorByPoint.get(d)) || ir.getCssColor(ir.resolveColor(d.hwKey as string)); @@ -3511,9 +3910,20 @@ const ScatterGraph = React.memo( zoomGroup.selectAll('.unofficial-overlay-pt').each(function (d) { const visible = ir.isOverlayPointVisible(d); d3.select(this) - .style('opacity', visible ? 1 : 0) + .style('opacity', visible ? pointOpacityForVariant(d) : 0) .style('pointer-events', visible ? 'auto' : 'none'); }); + // Overlay rooflines are only drawn for active overlay hardware; a + // comparison row hidden from the legend is the one visibility toggle + // they answer to here. + zoomGroup.selectAll('.overlay-roofline-path').each(function () { + const roofline = d3.select(this); + if (ir.hiddenPowerVariants.has(this.dataset.powerVariant ?? '')) { + roofline.style('opacity', 0); + } else { + roofline.style('opacity', null); + } + }); // Rooflines: visibility and solid-stroke recolor as direct writes. Keep // gradient url references intact and never touch animated path geometry. @@ -3523,7 +3933,9 @@ const ScatterGraph = React.memo( if (!hw || !precision) return; const roofline = d3.select(this); const visible = - ir.effectiveActiveHwTypes.has(hw) && ir.selectedPrecisions.includes(precision); + ir.effectiveActiveHwTypes.has(hw) && + ir.selectedPrecisions.includes(precision) && + !ir.hiddenPowerVariants.has(this.dataset.powerVariant ?? ''); roofline.style('opacity', visible ? 1 : 0); const stroke = roofline.attr('stroke'); if (stroke && !stroke.startsWith('url(')) { @@ -3557,6 +3969,7 @@ const ScatterGraph = React.memo( (this as SVGGElement).dataset, ir.effectiveActiveHwTypes, ir.selectedPrecisions, + ir.activeOverlayHwTypes, ); }); }, [ @@ -3715,7 +4128,9 @@ const ScatterGraph = React.memo( // brings a hidden ruler back); curves whose paths left the DOM // entirely are truly gone from the data, so prune each ruler (and the // draft) that references one. `prunePerfRulers` bails out with the - // same reference when nothing changed. + // same reference when nothing changed. Share-link rulers still + // pending are not state yet, so prune cannot touch them; drawPerfRuler + // above committed those whose curves now exist. setPerfRulerState((prev) => prunePerfRulers(prev, (cls) => !display.zoomGroup.select(`.${CSS.escape(cls)}`).empty()), ); @@ -3977,6 +4392,9 @@ const ScatterGraph = React.memo( ) : null, })), + // Comparison series (`i_pcompare`): one dash-swatch row per + // boundary / role, toggling that series across every hardware. + ...powerVariantLegendItems, ]} disableActiveSort={false} isLegendExpanded={isLegendExpanded} @@ -4141,7 +4559,10 @@ const ScatterGraph = React.memo( // the pre-paint decoration effect then removes the rulers // and the curve hit strokes before the next frame (no // lingering lines after toggle-off). - if (!checked) setPerfRulerState(clearPerfRulers); + if (!checked) { + setPerfRulerState(clearPerfRulers); + persistedRulers?.discardPending(); + } track('latency_perf_ruler_toggled', { enabled: checked }); }, }, @@ -4189,6 +4610,7 @@ const ScatterGraph = React.memo( count: perfRulerState.rulers.length, }); setPerfRulerState(clearPerfRulers); + persistedRulers?.discardPending(); }, }, ] diff --git a/packages/app/src/components/inference/ui/line-label-visibility.test.ts b/packages/app/src/components/inference/ui/line-label-visibility.test.ts index 80e17b049..dab9546f6 100644 --- a/packages/app/src/components/inference/ui/line-label-visibility.test.ts +++ b/packages/app/src/components/inference/ui/line-label-visibility.test.ts @@ -53,6 +53,64 @@ describe('labelOpacityForActiveState', () => { }); }); +describe('labelOpacityForActiveState with ?unofficialrun= overlays', () => { + const official = new Set(['gb200_dynamo-sglang']); + const overlay = new Set(['gb200_dynamo-sglang', 'gb300_dynamo-sglang']); + const precisions = ['fp8']; + + it('keeps an overlay label whose hardware is active only in the overlay legend', () => { + expect( + labelOpacityForActiveState( + { + hwKey: 'gb300_dynamo-sglang', + lineKey: 'overlay-gb300_dynamo-sglang_fp8_run0', + visible: '1', + }, + official, + precisions, + overlay, + ), + ).toBe(1); + }); + + it('hides an overlay label when its overlay hardware row is off, even if the official row is on', () => { + expect( + labelOpacityForActiveState( + { + hwKey: 'gb200_dynamo-sglang', + lineKey: 'overlay-gb200_dynamo-sglang_fp8_run1', + visible: '1', + }, + official, + precisions, + new Set(['gb300_dynamo-sglang']), + ), + ).toBe(0); + }); + + it('leaves official labels on the official set and falls back to it without an overlay set', () => { + expect( + labelOpacityForActiveState( + { hwKey: 'gb300_dynamo-sglang', lineKey: 'gb300_dynamo-sglang_fp8', visible: '1' }, + official, + precisions, + overlay, + ), + ).toBe(0); + expect( + labelOpacityForActiveState( + { + hwKey: 'gb300_dynamo-sglang', + lineKey: 'overlay-gb300_dynamo-sglang_fp8_run0', + visible: '1', + }, + official, + precisions, + ), + ).toBe(0); + }); +}); + describe('labelOpacityForHover', () => { it('lights up the kept label for the hovered hardware', () => { expect(labelOpacityForHover({ hwKey: 'b300_sglang', visible: '1' }, 'b300_sglang')).toBe(1); diff --git a/packages/app/src/components/inference/ui/line-label-visibility.ts b/packages/app/src/components/inference/ui/line-label-visibility.ts index a8b34b810..9dfa8e878 100644 --- a/packages/app/src/components/inference/ui/line-label-visibility.ts +++ b/packages/app/src/components/inference/ui/line-label-visibility.ts @@ -25,6 +25,8 @@ export interface LabelAttrs { /** `data-hw-key` — base hardware key, shared across a hw's curves. */ hwKey?: string; + /** `data-line-key` — `overlay-…` marks an unofficial-run curve's label. */ + lineKey?: string; /** `data-precision` — set on parallelism labels, absent on line labels. */ precision?: string; /** `data-visible` — `'1'`/`'0'`; only line labels set this. */ @@ -50,16 +52,24 @@ export const labelOpacityForHover = (attrs: LabelAttrs, hoveredHwKey: string): 0 * filter-change sync effect. Line labels (no precision) show when their * hardware is active **and** the render kept them; parallelism labels show when * their hardware is active and their precision is selected. + * + * An `?unofficialrun=` overlay curve answers to the overlay legend rows, not + * the official ones: its label follows `activeOverlayHwTypes`, so soloing an + * official hardware no longer hides the overlay pills of every other hardware + * (and hiding an official row keeps its overlay twin labelled). */ export const labelOpacityForActiveState = ( attrs: LabelAttrs, activeHwTypes: ReadonlySet, selectedPrecisions: readonly string[], + activeOverlayHwTypes?: ReadonlySet, ): 0 | 1 => { const { hwKey, precision } = attrs; if (!hwKey) return 0; + const isOverlay = attrs.lineKey?.startsWith('overlay-') ?? false; + const active = isOverlay && activeOverlayHwTypes ? activeOverlayHwTypes : activeHwTypes; if (!precision) { - return activeHwTypes.has(hwKey) && renderKept(attrs) ? 1 : 0; + return active.has(hwKey) && renderKept(attrs) ? 1 : 0; } - return activeHwTypes.has(hwKey) && selectedPrecisions.includes(precision) ? 1 : 0; + return active.has(hwKey) && selectedPrecisions.includes(precision) ? 1 : 0; }; diff --git a/packages/app/src/components/inference/utils.ts b/packages/app/src/components/inference/utils.ts index 5284aac3b..1794cd585 100644 --- a/packages/app/src/components/inference/utils.ts +++ b/packages/app/src/components/inference/utils.ts @@ -8,8 +8,15 @@ import { getGpuSpecs, type TcoBasis } from '@/lib/constants'; import chartDefinitions from '@/components/inference/metric-registry'; import { resolveXAxisField } from '@/components/inference/utils/resolveXAxisField'; import { remapInferencePoint } from '@/lib/chart-utils'; +import { expandPowerCompareSeries } from '@/components/inference/utils/power-compare'; -import type { ChartDefinition, ClippedInferenceData, InferenceData, YAxisMetricKey } from './types'; +import type { + ChartDefinition, + ClippedInferenceData, + InferenceData, + PowerCompare, + YAxisMetricKey, +} from './types'; import type { XAxisMode } from './hooks/useChartData'; /** @@ -145,6 +152,7 @@ export function processOverlayChartData( selectedXAxisMode?: XAxisMode; restrictToNormalizedFrontier?: boolean; tcoBasis?: TcoBasis; + powerCompare?: PowerCompare; }, ): InferenceData[] { return processOverlayChartDataWithClipping( @@ -171,6 +179,8 @@ export function processOverlayChartDataWithClipping( selectedXAxisMode?: XAxisMode; restrictToNormalizedFrontier?: boolean; tcoBasis?: TcoBasis; + /** Sibling boundary / role series, mirroring the official path in useChartData. */ + powerCompare?: PowerCompare; }, ): ProcessedChartData { const chartDef = (chartDefinitions as ChartDefinition[]).find((d) => d.chartType === chartType); @@ -222,9 +232,13 @@ export function processOverlayChartDataWithClipping( // for the natural axis and for agentic (long TTFTs are normal there). const isTtftX = xAxisField.endsWith('_ttft'); - const processedData = sourceData - .filter((d) => metricKey in d) - .map((d) => remapInferencePoint(d, metricKey, xAxisField)); + const processedData = expandPowerCompareSeries( + sourceData + .filter((d) => metricKey in d) + .map((d) => remapInferencePoint(d, metricKey, xAxisField)), + selectedYAxisMetric, + options?.powerCompare ?? 'none', + ); // The normalized metric is derived from persisted request traces, which an // unofficial overlay does not have. An all-false canonical stamp prevents a diff --git a/packages/app/src/components/inference/utils/best-series-per-sku.ts b/packages/app/src/components/inference/utils/best-series-per-sku.ts index 6799b9631..2a35f0b9c 100644 --- a/packages/app/src/components/inference/utils/best-series-per-sku.ts +++ b/packages/app/src/components/inference/utils/best-series-per-sku.ts @@ -40,7 +40,9 @@ export function bestSeriesPerSku(points: InferenceData[], direction: Direction): const bySku = new Map>(); const featured = new Set(); for (const point of points) { - if (!isFrontierEligible(point) || !Number.isFinite(point.y)) continue; + // Comparison clones re-plot the same configs at another boundary or role; + // the best series per SKU is judged on the selected metric alone. + if (point.powerVariant || !isFrontierEligible(point) || !Number.isFinite(point.y)) continue; const sku = baseSku(point); const key = String(point.hwKey); if (point.framework === 'tilert') featured.add(key); diff --git a/packages/app/src/components/inference/utils/point-identity.test.ts b/packages/app/src/components/inference/utils/point-identity.test.ts index 26d71f322..a53326170 100644 --- a/packages/app/src/components/inference/utils/point-identity.test.ts +++ b/packages/app/src/components/inference/utils/point-identity.test.ts @@ -2,7 +2,12 @@ import { describe, expect, it } from 'vitest'; import type { InferenceData } from '@/components/inference/types'; -import { scatterPointConfigId, scatterPointJoinId } from './point-identity'; +import { + parseScatterSeriesKey, + scatterPointConfigId, + scatterPointJoinId, + scatterSeriesKey, +} from './point-identity'; const point = (overrides: Partial): InferenceData => ({ @@ -85,3 +90,26 @@ describe('scatterPointConfigId', () => { expect(scatterPointJoinId(undated, true)).toBe(scatterPointConfigId(undated)); }); }); + +describe('scatterSeriesKey', () => { + it('keeps comparison clones in their own series and parses the key back', () => { + const base = point({}); + const clone = point({ powerVariant: { kind: 'basis', id: 'gpu-provisioned' } }); + expect(scatterSeriesKey(base)).toBe('h200_vllm_fp8'); + expect(scatterSeriesKey(clone)).toBe('h200_vllm_fp8-v-gpu-provisioned'); + expect(parseScatterSeriesKey('h200_vllm_fp8')).toEqual({ + hw: 'h200_vllm', + precision: 'fp8', + variant: null, + }); + expect(parseScatterSeriesKey('b200_sglang_mtp_fp4-v-utility-modeled')).toEqual({ + hw: 'b200_sglang_mtp', + precision: 'fp4', + variant: 'utility-modeled', + }); + // The variant is point identity too, so a clone never replaces its base in a D3 join. + expect(scatterPointConfigId(clone)).toBe( + `${scatterPointConfigId(base)}|variant-gpu-provisioned`, + ); + }); +}); diff --git a/packages/app/src/components/inference/utils/point-identity.ts b/packages/app/src/components/inference/utils/point-identity.ts index 2c975593e..0cc8bbed8 100644 --- a/packages/app/src/components/inference/utils/point-identity.ts +++ b/packages/app/src/components/inference/utils/point-identity.ts @@ -23,9 +23,47 @@ export function scatterPointConfigId(point: InferenceData): string { // Agentic series omit spec decoding from hwKey so one curve can mix methods. // It remains point identity to avoid collapsing overlapping MTP/STP results. key += agenticSpecDecodingKeySuffix(point); + // Comparison clones share every config field with their base point. + if (point.powerVariant) key += `|variant-${point.powerVariant.id}`; return key; } +/** + * Comparison-series suffix inside a scatter series key. Letters, digits and + * dashes only, so the key stays a valid CSS class token (the perf ruler and + * `i_rulers` address rooflines by class) and needs no escaping. + */ +const SERIES_VARIANT_DELIMITER = '-v-'; + +/** + * Identity of one drawn series: hardware key, precision and, on a power + * comparison, the boundary or role variant. Rooflines, frontiers, line labels + * and the perf ruler all key on this string. + */ +export function scatterSeriesKey( + point: Pick, +): string { + const base = `${point.hwKey}_${point.precision}`; + return point.powerVariant ? `${base}${SERIES_VARIANT_DELIMITER}${point.powerVariant.id}` : base; +} + +export interface ScatterSeriesIdentity { + hw: string; + precision: string; + /** Comparison variant id (`gpu-provisioned`, `prefill`, …) or null for the base series. */ + variant: string | null; +} + +/** Inverse of `scatterSeriesKey`; hardware keys may themselves contain underscores. */ +export function parseScatterSeriesKey(key: string): ScatterSeriesIdentity { + const delimiter = key.indexOf(SERIES_VARIANT_DELIMITER); + const core = delimiter === -1 ? key : key.slice(0, delimiter); + const variant = delimiter === -1 ? null : key.slice(delimiter + SERIES_VARIANT_DELIMITER.length); + const parts = core.split('_'); + const precision = parts.pop() ?? ''; + return { hw: parts.join('_'), precision, variant }; +} + /** * Stable D3 join key for an official scatter point. * diff --git a/packages/app/src/components/inference/utils/power-compare.test.ts b/packages/app/src/components/inference/utils/power-compare.test.ts new file mode 100644 index 000000000..6b7c76153 --- /dev/null +++ b/packages/app/src/components/inference/utils/power-compare.test.ts @@ -0,0 +1,295 @@ +import { describe, expect, it } from 'vitest'; + +import type { + InferenceData, + PowerBasis, + PowerRole, + PowerVariant, +} from '@/components/inference/types'; + +import { + expandPowerCompareSeries, + flatSeriesValue, + formatWatts, + inferPowerCompare, + lineLabelHardwareKey, + lineLabelSeriesId, + metricPlotsWatts, + parsePowerCompare, + powerCompareAvailable, + powerCompareBase, + powerCompareVariants, + powerLineLabel, + powerLineLabelSuffix, + powerSeriesLabel, + powerVariantDash, + powerVariantLabel, + powerVariantShortLabel, + powerVariantsInData, +} from './power-compare'; + +const metric = (y: number) => ({ y, roof: false }); + +function point(overrides: Partial = {}): InferenceData { + return { + x: 50, + y: 600, + hwKey: 'b200_sglang', + precision: 'fp8', + tp: 8, + conc: 64, + disagg: true, + measuredAvgPower: metric(600), + measuredPrefillAvgPower: metric(400), + measuredDecodeAvgPower: metric(700), + gpuProvisionedWatts: metric(1000), + utilityProvisionedWatts: metric(1710), + measuredJPerOutputToken: metric(7.9), + measuredDecodeJPerOutputToken: metric(6), + reconstructedPrefillJPerOutputToken: metric(1.975), + gpuProvisionedJPerOutputToken: metric(12), + ...overrides, + } as InferenceData; +} + +describe('powerCompareVariants', () => { + it('adds the other three boundaries on the whole-deployment average watts and J per output token', () => { + expect(powerCompareVariants('y_measuredAvgPower', 'boundaries')).toEqual([ + { variant: { kind: 'basis', id: 'gpu-provisioned' }, field: 'gpuProvisionedWatts' }, + { variant: { kind: 'basis', id: 'utility-provisioned' }, field: 'utilityProvisionedWatts' }, + { variant: { kind: 'basis', id: 'utility-modeled' }, field: 'utilityModeledWatts' }, + ]); + // Selecting a derived boundary makes GPU measured one of the siblings. + expect( + powerCompareVariants('y_utilityProvisionedJPerOutputToken', 'boundaries').map( + (series) => series.field, + ), + ).toEqual([ + 'measuredJPerOutputToken', + 'gpuProvisionedJPerOutputToken', + 'utilityModeledJPerOutputToken', + ]); + expect(powerCompareBase('y_gpuProvisionedWatts', 'boundaries')).toEqual({ + kind: 'basis', + id: 'gpu-provisioned', + }); + }); + + it('adds the other worker roles, carrying prefill energy onto the output-token axis', () => { + expect(powerCompareVariants('y_measuredAvgPower', 'roles')).toEqual([ + { variant: { kind: 'role', id: 'prefill' }, field: 'measuredPrefillAvgPower' }, + { variant: { kind: 'role', id: 'decode' }, field: 'measuredDecodeAvgPower' }, + ]); + expect(powerCompareVariants('y_measuredDecodeAvgPower', 'roles')).toEqual([ + { variant: { kind: 'role', id: 'all' }, field: 'measuredAvgPower' }, + { variant: { kind: 'role', id: 'prefill' }, field: 'measuredPrefillAvgPower' }, + ]); + expect(powerCompareVariants('y_measuredJPerOutputToken', 'roles')).toEqual([ + { variant: { kind: 'role', id: 'prefill' }, field: 'reconstructedPrefillJPerOutputToken' }, + { variant: { kind: 'role', id: 'decode' }, field: 'measuredDecodeJPerOutputToken' }, + ]); + expect(powerCompareBase('y_measuredDecodeJPerOutputToken', 'roles')).toEqual({ + kind: 'role', + id: 'decode', + }); + }); + + it('offers nothing where the metric has no common axis for the siblings', () => { + for (const [key, mode] of [ + ['y_measuredP90Power', 'boundaries'], + ['y_measuredPowerPercentTdp', 'boundaries'], + ['y_measuredPowerTimeline', 'roles'], + ['y_measuredJPerInputToken', 'boundaries'], + ['y_measuredPrefillJPerInputToken', 'roles'], + ['y_measuredJPerSuccessfulQuery', 'roles'], + ['y_gpuProvisionedWatts', 'roles'], + ['y_tpPerGpu', 'boundaries'], + ] as const) { + expect(powerCompareVariants(key, mode)).toEqual([]); + expect(powerCompareBase(key, mode)).toBeNull(); + expect(powerCompareAvailable(key, mode)).toBe(false); + } + expect(powerCompareAvailable('y_measuredP90Power', 'none')).toBe(true); + expect(powerCompareVariants('y_measuredAvgPower', 'none')).toEqual([]); + }); +}); + +describe('expandPowerCompareSeries', () => { + it('clones each base point once per sibling that has a value and leaves the base untouched', () => { + const base = point(); + const expanded = expandPowerCompareSeries([base], 'y_measuredAvgPower', 'boundaries'); + expect(expanded[0]).toBe(base); + expect(expanded.slice(1).map((p) => [p.powerVariant?.id, p.y])).toEqual([ + ['gpu-provisioned', 1000], + ['utility-provisioned', 1710], + ]); + // No modeled watts on this point → no utility-modeled clone, never a 0. + expect(expanded.some((p) => p.powerVariant?.id === 'utility-modeled')).toBe(false); + expect(expanded.slice(1).every((p) => p.hwKey === 'b200_sglang' && p.x === 50)).toBe(true); + }); + + it('stacks the reconstructed prefill energy against the decode pool on the energy axis', () => { + const expanded = expandPowerCompareSeries( + [point({ y: 7.9 })], + 'y_measuredJPerOutputToken', + 'roles', + ); + expect(expanded.map((p) => [p.powerVariant?.id ?? 'base', p.y])).toEqual([ + ['base', 7.9], + ['prefill', 1.975], + ['decode', 6], + ]); + expect(inferPowerCompare(expanded)).toBe('roles'); + expect(powerVariantsInData(expanded, 'y_measuredJPerOutputToken')).toEqual([ + { kind: 'role', id: 'all' }, + { kind: 'role', id: 'prefill' }, + { kind: 'role', id: 'decode' }, + ]); + }); + + it('returns the points unchanged when the mode is off or inapplicable', () => { + const base = point(); + expect(expandPowerCompareSeries([base], 'y_measuredAvgPower', 'none')).toEqual([base]); + expect(expandPowerCompareSeries([base], 'y_measuredP90Power', 'boundaries')).toEqual([base]); + expect(inferPowerCompare([base])).toBe('none'); + expect(powerVariantsInData([base], 'y_measuredAvgPower')).toEqual([]); + }); +}); + +describe('labels, dashes and URL values', () => { + it('parses only the two comparison modes from the URL', () => { + expect(parsePowerCompare('boundaries')).toBe('boundaries'); + expect(parsePowerCompare('roles')).toBe('roles'); + for (const value of ['', 'none', 'basis', null, undefined]) { + expect(parsePowerCompare(value)).toBe('none'); + } + }); + + it('names series in both locales and keeps the base series solid', () => { + expect(powerVariantLabel({ kind: 'basis', id: 'gpu-provisioned' }, 'en')).toBe( + 'GPU provisioned (TDP)', + ); + expect(powerVariantLabel({ kind: 'role', id: 'prefill' }, 'zh')).toBe('预填充 GPU'); + expect(powerVariantDash(undefined)).toBe(''); + expect(powerVariantDash({ kind: 'role', id: 'all' })).toBe(''); + expect(powerVariantDash({ kind: 'basis', id: 'gpu-provisioned' })).not.toBe(''); + expect(powerVariantDash({ kind: 'role', id: 'decode' })).not.toBe( + powerVariantDash({ kind: 'role', id: 'prefill' }), + ); + }); + + it('labels a base row by the selected metric and a clone by its variant', () => { + expect(powerSeriesLabel({}, 'y_measuredAvgPower', 'boundaries', 'en')).toBe('GPU measured'); + expect( + powerSeriesLabel( + { powerVariant: { kind: 'basis', id: 'utility-modeled' } }, + 'y_measuredAvgPower', + 'boundaries', + 'zh', + ), + ).toBe('数据中心建模(含 PUE)'); + expect(powerSeriesLabel({}, 'y_measuredAvgPower', 'none', 'en')).toBe(''); + }); +}); + +const basis = (id: PowerBasis): PowerVariant => ({ kind: 'basis', id }); +const role = (id: PowerRole): PowerVariant => ({ kind: 'role', id }); + +describe('line labels', () => { + it('names every variant briefly in both locales', () => { + const cases: [PowerVariant, string, string][] = [ + [basis('gpu-measured'), 'Measured', '实测'], + [basis('gpu-provisioned'), 'TDP', 'TDP'], + [basis('utility-provisioned'), 'All-in', '全站'], + [basis('utility-modeled'), 'PUE modeled', 'PUE 建模'], + [role('all'), 'All GPUs', '全部 GPU'], + [role('prefill'), 'Prefill GPUs', '预填充 GPU'], + [role('decode'), 'Decode GPUs', '解码 GPU'], + ]; + for (const [variant, en, zh] of cases) { + expect(powerVariantShortLabel(variant, 'en')).toBe(en); + expect(powerVariantShortLabel(variant, 'zh')).toBe(zh); + } + }); + + it('formats watts as integer W below 1 kW and trimmed two-decimal kW above', () => { + expect(formatWatts(700)).toBe('700 W'); + expect(formatWatts(999.6)).toBe('1000 W'); + expect(formatWatts(1000)).toBe('1 kW'); + expect(formatWatts(1200)).toBe('1.2 kW'); + expect(formatWatts(1370)).toBe('1.37 kW'); + expect(formatWatts(1400)).toBe('1.4 kW'); + expect(formatWatts(1714.5)).toBe('1.71 kW'); + expect(formatWatts(19200)).toBe('19.2 kW'); + expect(formatWatts(100000)).toBe('100 kW'); + }); + + it('leaves the base label alone and suffixes a sibling with its variant and flat watts', () => { + const label = 'B300 (SGLang)'; + const tdp = basis('gpu-provisioned'); + expect(powerLineLabel(label, undefined, { isBase: true, locale: 'en' })).toBe(label); + expect(powerLineLabel(label, null, { isBase: false, locale: 'en' })).toBe(label); + // A variant that is itself the base series keeps the plain label. + expect(powerLineLabel(label, tdp, { isBase: true, locale: 'en', flatWatts: 1200 })).toBe(label); + expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en' })).toBe('B300 (SGLang) · TDP'); + expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en', flatWatts: 700 })).toBe( + 'B300 (SGLang) · TDP 700 W', + ); + expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en', flatWatts: 1370 })).toBe( + 'B300 (SGLang) · TDP 1.37 kW', + ); + expect( + powerLineLabel(label, basis('utility-provisioned'), { + isBase: false, + locale: 'zh', + flatWatts: 19200, + }), + ).toBe('B300 (SGLang) · 全站 19.2 kW'); + // Non-finite or null watts drop the value, never print NaN. + expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en', flatWatts: null })).toBe( + 'B300 (SGLang) · TDP', + ); + expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en', flatWatts: NaN })).toBe( + 'B300 (SGLang) · TDP', + ); + expect(powerLineLabel(label, role('decode'), { isBase: false, locale: 'en' })).toBe( + 'B300 (SGLang) · Decode GPUs', + ); + // The suffix alone is what the renderer splits into its own text segment. + expect(powerLineLabelSuffix(role('prefill'), { isBase: false, locale: 'zh' })).toBe( + ' · 预填充 GPU', + ); + expect(powerLineLabelSuffix(role('prefill'), { isBase: true, locale: 'zh' })).toBe(''); + }); + + it('detects a flat series within the relative tolerance', () => { + expect(flatSeriesValue([1200, 1200, 1200])).toBe(1200); + expect(flatSeriesValue([1000, 1004, 996])).toBe(1000); + expect(flatSeriesValue([1000, 1005])).toBe(1000); + expect(flatSeriesValue([1000, 1005.01])).toBeNull(); + expect(flatSeriesValue([1000, 1005.01], 0.01)).toBe(1000); + expect(flatSeriesValue([600, 640, 710])).toBeNull(); + expect(flatSeriesValue([])).toBeNull(); + expect(flatSeriesValue([NaN, Infinity])).toBeNull(); + expect(flatSeriesValue([NaN, 1200, 1200])).toBe(1200); + expect(flatSeriesValue([1200])).toBe(1200); + }); + + it('keeps the hardware key as the base series id and recovers it from a sibling id', () => { + const tdp = basis('gpu-provisioned'); + expect(lineLabelSeriesId('b200_sglang', undefined, true)).toBe('b200_sglang'); + expect(lineLabelSeriesId('b200_sglang', tdp, true)).toBe('b200_sglang'); + expect(lineLabelSeriesId('b200_sglang', tdp, false)).toBe('b200_sglang::gpu-provisioned'); + expect(lineLabelHardwareKey('b200_sglang::gpu-provisioned')).toBe('b200_sglang'); + expect(lineLabelHardwareKey('b200_sglang')).toBe('b200_sglang'); + }); + + it('only lets a watts axis state a flat boundary value', () => { + expect(metricPlotsWatts('y_measuredAvgPower')).toBe(true); + expect(metricPlotsWatts('y_gpuProvisionedWatts')).toBe(true); + expect(metricPlotsWatts('y_measuredDecodeAvgPower')).toBe(true); + expect(metricPlotsWatts('y_measuredPowerPercentTdp')).toBe(false); + expect(metricPlotsWatts('y_measuredJPerOutputToken')).toBe(false); + expect(metricPlotsWatts('y_tpPerGpu')).toBe(false); + }); +}); diff --git a/packages/app/src/components/inference/utils/power-compare.ts b/packages/app/src/components/inference/utils/power-compare.ts new file mode 100644 index 000000000..19300b8cd --- /dev/null +++ b/packages/app/src/components/inference/utils/power-compare.ts @@ -0,0 +1,344 @@ +/** + * Comparison series for the gated measured-power charts (`i_pcompare`). + * + * The selected metric stays the chart's base series. A comparison adds sibling + * series drawn from the SAME points — one per other power boundary + * (`boundaries`, PowerX Figures 2/3) or per worker role (`roles`, Figures 6/7) + * — so ScatterGraph can draw them with the hardware's colour and a per-variant + * dash through its ordinary series pipeline (frontier, Optimal Only, tooltip, + * table, CSV, `?unofficialrun=` overlay). A variant point is a clone of its + * base point with `y` remapped and `powerVariant` set; base points are left + * untouched so a chart without comparison is byte-identical to before. + * + * Variants exist only where the metric names a whole-deployment average + * (W per chip) or J per output token, because those are the only quantities + * every boundary and role publishes on a common axis; everywhere else the + * comparison yields no series and the control says so. + */ +import type { Locale } from '@/lib/i18n'; +import { + POWER_BASES, + POWER_BASIS_FIELDS, + POWER_BASIS_LABELS, + type PowerBasis, +} from '@/lib/power-basis'; + +import { getMeasuredMetricConfig } from '../measured-metric-config'; +import type { InferenceData, PowerCompare, PowerRole, PowerVariant } from '../types'; +import { reconstructedRoleEnergy } from './role-energy'; + +export const POWER_COMPARE_MODES = [ + 'none', + 'boundaries', + 'roles', +] as const satisfies readonly PowerCompare[]; + +export function parsePowerCompare(value: string | null | undefined): PowerCompare { + return value === 'boundaries' || value === 'roles' ? value : 'none'; +} + +/** Point fields a comparison series can plot; each is a `{ y, roof }` pair. */ +export type PowerSeriesField = keyof Pick< + InferenceData, + | 'measuredAvgPower' + | 'measuredPrefillAvgPower' + | 'measuredDecodeAvgPower' + | 'measuredJPerOutputToken' + | 'measuredDecodeJPerOutputToken' + | 'reconstructedPrefillJPerOutputToken' + | 'gpuProvisionedWatts' + | 'gpuProvisionedJPerOutputToken' + | 'utilityProvisionedWatts' + | 'utilityProvisionedJPerOutputToken' + | 'utilityModeledWatts' + | 'utilityModeledJPerOutputToken' +>; + +export interface PowerCompareSeries { + variant: PowerVariant; + field: PowerSeriesField; +} + +const ROLES = ['all', 'prefill', 'decode'] as const satisfies readonly PowerRole[]; + +const ROLE_WATT_FIELDS: Record = { + all: 'measuredAvgPower', + prefill: 'measuredPrefillAvgPower', + decode: 'measuredDecodeAvgPower', +}; +// Energy per output token per role. The prefill pool's own figure is per input +// token; `reconstructedPrefillJPerOutputToken` carries it onto the output-token +// axis (utils/role-energy.ts). +const ROLE_ENERGY_FIELDS: Record = { + all: 'measuredJPerOutputToken', + prefill: 'reconstructedPrefillJPerOutputToken', + decode: 'measuredDecodeJPerOutputToken', +}; + +const ROLE_LABELS: Record = { + all: { en: 'All GPUs', zh: '全部 GPU' }, + prefill: { en: 'Prefill GPUs', zh: '预填充 GPU' }, + decode: { en: 'Decode GPUs', zh: '解码 GPU' }, +}; + +/** SVG dash per variant; the base series and `all` stay solid. */ +const VARIANT_DASH: Record = { + 'gpu-measured': '', + 'gpu-provisioned': '8 4', + 'utility-provisioned': '3 3', + 'utility-modeled': '10 3 2 3', + all: '', + prefill: '7 3', + decode: '2 3', +}; + +/** + * The quantity a metric key plots on the comparison's common axis, or null + * when the key is not a whole-deployment average W/chip or J per output token. + */ +function comparableQuantity(metric: string): 'watts' | 'energy' | null { + const config = getMeasuredMetricConfig(metric); + if (!config) return null; + if (config.family === 'power') { + return config.scope === 'all' && config.statistic === 'average' && config.display === 'watts' + ? 'watts' + : null; + } + return config.scope === 'all' && config.denominator === 'output' && config.unit === 'joules' + ? 'energy' + : null; +} + +/** + * The base series' own identity under a comparison — the boundary or role the + * selected metric already plots — or null when the metric admits no comparison + * of that kind. + */ +export function powerCompareBase(metric: string, mode: PowerCompare): PowerVariant | null { + const config = getMeasuredMetricConfig(metric); + if (!config || mode === 'none') return null; + if (mode === 'boundaries') { + return comparableQuantity(metric) ? { kind: 'basis', id: config.basis } : null; + } + if (config.basis !== 'gpu-measured') return null; + if (config.family === 'power') { + return config.statistic === 'average' && config.display === 'watts' + ? { kind: 'role', id: config.scope } + : null; + } + // Role energy compares J per output token; a prefill J per input token axis + // has no decode counterpart. + return config.denominator === 'output' && config.unit === 'joules' && config.scope !== 'prefill' + ? { kind: 'role', id: config.scope } + : null; +} + +/** The sibling series a comparison adds to the selected metric (never the base itself). */ +export function powerCompareVariants(metric: string, mode: PowerCompare): PowerCompareSeries[] { + const base = powerCompareBase(metric, mode); + const config = getMeasuredMetricConfig(metric); + if (!base || !config) return []; + if (base.kind === 'basis') { + const quantity = comparableQuantity(metric); + if (!quantity) return []; + return POWER_BASES.filter((basis) => basis !== base.id).map((basis) => ({ + variant: { kind: 'basis', id: basis }, + field: + basis === 'gpu-measured' + ? quantity === 'watts' + ? 'measuredAvgPower' + : 'measuredJPerOutputToken' + : POWER_BASIS_FIELDS[basis][quantity], + })); + } + const fields = config.family === 'power' ? ROLE_WATT_FIELDS : ROLE_ENERGY_FIELDS; + return ROLES.filter((role) => role !== base.id).map((role) => ({ + variant: { kind: 'role', id: role }, + field: fields[role], + })); +} + +/** Whether choosing `mode` on `metric` draws anything. `none` is always available. */ +export function powerCompareAvailable(metric: string, mode: PowerCompare): boolean { + return mode === 'none' || powerCompareVariants(metric, mode).length > 0; +} + +/** + * Appends one clone per comparison series to `points` (already remapped onto + * the selected metric). A point lacking a variant's field contributes nothing + * to that series — never a 0 — so the availability rules of each boundary and + * role carry through unchanged. + */ +export function expandPowerCompareSeries( + points: readonly InferenceData[], + metric: string, + mode: PowerCompare, +): InferenceData[] { + const series = powerCompareVariants(metric, mode); + if (series.length === 0) return [...points]; + const result: InferenceData[] = [...points]; + for (const { variant, field } of series) { + for (const point of points) { + const value = point[field]; + if (!value || !Number.isFinite(value.y)) continue; + result.push({ ...point, y: value.y, roof: value.roof, powerVariant: variant }); + } + } + return result; +} + +/** Comparison mode implied by the variants present in a rendered point set. */ +export function inferPowerCompare(points: readonly InferenceData[]): PowerCompare { + for (const point of points) { + if (point.powerVariant?.kind === 'basis') return 'boundaries'; + if (point.powerVariant?.kind === 'role') return 'roles'; + } + return 'none'; +} + +/** Distinct variants in draw order: the base first, then siblings in canonical order. */ +export function powerVariantsInData( + points: readonly InferenceData[], + metric: string, +): PowerVariant[] { + const mode = inferPowerCompare(points); + if (mode === 'none') return []; + const present = new Set(points.map((point) => point.powerVariant?.id).filter(Boolean)); + const base = powerCompareBase(metric, mode); + const ordered: PowerVariant[] = + mode === 'boundaries' + ? POWER_BASES.map((id) => ({ kind: 'basis', id })) + : ROLES.map((id) => ({ kind: 'role', id })); + return ordered.filter((variant) => variant.id === base?.id || present.has(variant.id)); +} + +export function powerVariantId(variant: PowerVariant | null | undefined): string { + return variant?.id ?? ''; +} + +export function powerVariantDash(variant: PowerVariant | null | undefined): string { + return variant ? VARIANT_DASH[variant.id] : ''; +} + +export function powerVariantLabel(variant: PowerVariant, locale: Locale): string { + return variant.kind === 'basis' + ? POWER_BASIS_LABELS[variant.id][locale] + : ROLE_LABELS[variant.id][locale]; +} + +/** Short boundary names for in-chart line labels; the legend keeps the full names. */ +const BASIS_SHORT_LABELS: Record = { + 'gpu-measured': { en: 'Measured', zh: '实测' }, + 'gpu-provisioned': { en: 'TDP', zh: 'TDP' }, + 'utility-provisioned': { en: 'All-in', zh: '全站' }, + 'utility-modeled': { en: 'PUE modeled', zh: 'PUE 建模' }, +}; + +/** Suffix text a line label carries for one comparison series. */ +export function powerVariantShortLabel(variant: PowerVariant, locale: Locale): string { + return variant.kind === 'basis' + ? BASIS_SHORT_LABELS[variant.id][locale] + : ROLE_LABELS[variant.id][locale]; +} + +/** `700 W`, `1.37 kW`, `19.2 kW`: kilowatts from 1000 W, at most two decimals, zeros trimmed. */ +export function formatWatts(watts: number): string { + if (Math.abs(watts) >= 1000) { + return `${(watts / 1000).toFixed(2).replace(/\.?0+$/u, '')} kW`; + } + return `${Math.round(watts)} W`; +} + +/** + * The value a series holds at every point, or null when it varies. A + * provisioned boundary (TDP, all-in) is one number per hardware, so its line + * label can state it instead of sending the reader to the axis. Non-finite + * values are ignored; an empty series is null. + */ +export function flatSeriesValue(values: readonly number[], relTolerance = 0.005): number | null { + const finite = values.filter((value) => Number.isFinite(value)); + if (finite.length === 0) return null; + const first = finite[0]; + const tolerance = Math.abs(first) * relTolerance; + return finite.every((value) => Math.abs(value - first) <= tolerance) ? first : null; +} + +export interface PowerLineLabelOptions { + /** The selected metric's own series keeps the plain hardware label. */ + isBase: boolean; + locale: Locale; + /** Shared watts of a flat series (`flatSeriesValue`), appended after the name. */ + flatWatts?: number | null; +} + +const LINE_LABEL_SUFFIX_SEPARATOR = ' · '; + +/** Suffix appended to a comparison sibling's line label; '' for the base series. */ +export function powerLineLabelSuffix( + variant: PowerVariant | null | undefined, + opts: PowerLineLabelOptions, +): string { + if (opts.isBase || !variant) return ''; + const watts = + typeof opts.flatWatts === 'number' && Number.isFinite(opts.flatWatts) + ? ` ${formatWatts(opts.flatWatts)}` + : ''; + return `${LINE_LABEL_SUFFIX_SEPARATOR}${powerVariantShortLabel(variant, opts.locale)}${watts}`; +} + +/** + * Line-label text for one drawn series: the hardware label alone for the base + * series, `
  • `; }; @@ -755,6 +855,7 @@ export const generateOverlayTooltipContent = (config: OverlayTooltipConfig): str ${tooltipLine(xLabel, fmt(d.x))} ${tooltipLine(yLabel, fmt(d.y))} ${powerTierHTML(d, selectedYAxisMetric, locale)} + ${powerVariantHTML(d, selectedYAxisMetric, locale)} ${modeledSystemPowerHTML(d, selectedYAxisMetric, isPinned, locale)} ${totalChipsHTML(d, selectedYAxisMetric, locale)} ${generateParallelismHTML(d, locale)} @@ -764,6 +865,14 @@ export const generateOverlayTooltipContent = (config: OverlayTooltipConfig): str ${powerWithheldHTML(d, locale)} ${generateAgenticHTML(d, locale)} ${generateWorkerPowerHTML(d, isPinned, locale)} + ${viewActionsHTML({ + isPinned, + hasTraceData: false, + hasLogData: false, + point: d, + powerTraceMetric: selectedYAxisMetric, + locale, + })} `; }; @@ -816,6 +925,7 @@ export const generateGPUGraphTooltipContent = (config: TooltipConfig): string => : '' } ${powerTierHTML(d, selectedYAxisMetric, locale)} + ${powerVariantHTML(d, selectedYAxisMetric, locale)} ${modeledSystemPowerHTML(d, selectedYAxisMetric, isPinned, locale)} ${totalChipsHTML(d, selectedYAxisMetric, locale)} ${generateParallelismHTML(d, locale)} @@ -826,7 +936,13 @@ export const generateGPUGraphTooltipContent = (config: TooltipConfig): string => ${generateAgenticHTML(d, locale)} ${generateWorkerPowerHTML(d, isPinned, locale)} ${runLinkHTML(runUrl, locale)} - ${viewActionsHTML(isPinned, Boolean(hasTrace), Boolean(hasLog), d.id, d.benchmark_type, locale)} + ${viewActionsHTML({ + isPinned, + hasTraceData: Boolean(hasTrace), + hasLogData: Boolean(hasLog), + point: d, + locale, + })} `; }; diff --git a/packages/app/src/components/tab-nav.tsx b/packages/app/src/components/tab-nav.tsx index fbc9066ad..a3ef979c4 100644 --- a/packages/app/src/components/tab-nav.tsx +++ b/packages/app/src/components/tab-nav.tsx @@ -47,6 +47,8 @@ const TAB_LABELS_EN: Record = { historical: 'Historical Trends', calculator: 'TCO Calculator', fleet: 'Fleet Lifecycle', + 'first-token': 'First-Token Limits', + 'cache-reuse': 'Prefix Cache Reuse', 'profit-estimator': 'Profit Estimator', 'profit-estimator-per-gigawatt': 'Profit Estimator per GW', reliability: 'Reliability', diff --git a/packages/app/src/lib/api-route-catalog.ts b/packages/app/src/lib/api-route-catalog.ts index 6f090ad9f..258ad7a6a 100644 --- a/packages/app/src/lib/api-route-catalog.ts +++ b/packages/app/src/lib/api-route-catalog.ts @@ -68,10 +68,10 @@ export const apiRouteCatalog = [ method: 'GET', classification: 'ui-artifact-read', exclusionReason: { - en: 'UI-only PowerX explorer read for one run: the ingest-time telemetry digest when stored, otherwise the live GPU metric artifacts. Its payload shape is not a stable public contract.', - zh: '仅供 PowerX 探索界面按 run 读取:已入库时返回 ingest 阶段生成的 telemetry 摘要,否则回退到实时 GPU 指标制品。其返回结构不是稳定的公开契约。', + en: 'UI-only PowerX read for one run: the ingest-time telemetry digest when stored, otherwise the live GPU telemetry artifacts (raw `gpu_metrics_*` rows, or `series=power` one-second buckets for the PowerX timeline, also cut per validation window from `power_audit_*` bundles). Its payload shape is not a stable public contract.', + zh: '仅供 PowerX 界面按 run 读取:已入库时返回 ingest 阶段生成的 telemetry 摘要,否则回退到实时 GPU 遥测制品(`gpu_metrics_*` 原始行,或供 PowerX 时间线使用的 `series=power` 一秒分桶数据,后者也会按验证窗口从 `power_audit_*` bundle 中切分得到)。其返回结构不是稳定的公开契约。', }, - sourceSha256: '01d604d77ea73e252f0934f88d14d0e229a803d1b9b1d72bd9756658c506b349', + sourceSha256: '4d387298f0311a91c319548119382646045fff6cccdf4e871adf89b39a6feee1', }, { source: 'src/app/api/openapi.json/route.ts', @@ -753,7 +753,10 @@ export const apiContractSourceDigests = [ // Reviewed again for the release-date corrections: values inside // MODEL_RELEASE_DATES only. No published model name, alias, or parameter enum // is touched, and no endpoint exposes a release date, so the docs stand. - sourceSha256: 'bb58d43160c2b83ce61e7c34326a6d316fd751e435c6819f1991ed69e4f1b45c', + // Reviewed for the Qwen3.8-27B addition (InferenceX#3260): two new DB keys + // and display names plus their release dates. No published parameter enum + // or endpoint changes, so the docs stand. + sourceSha256: 'af1053b2ae94b50de51153153dd7a7e268e50bde3e5900310e44f2d8baa86f87', reviewArea: { en: 'Published benchmark and TCO model names, aliases, and parameter enums.', zh: '已发布基准与 TCO 模型名称、别名和参数枚举。', diff --git a/packages/app/src/lib/benchmark-api-view.test.ts b/packages/app/src/lib/benchmark-api-view.test.ts index c74397c99..e840244cc 100644 --- a/packages/app/src/lib/benchmark-api-view.test.ts +++ b/packages/app/src/lib/benchmark-api-view.test.ts @@ -48,6 +48,36 @@ describe('toCalculatorBenchmarkRows', () => { ]); }); + it('keeps time to first token at the percentiles the calculator pages read', () => { + // The First-Token Limits page caps rows on TTFT through this same view; + // without these the page would have nothing to cap. p99 stays out like the + // other p99 latency metrics. + const [row] = toCalculatorBenchmarkRows( + [ + { + benchmark_type: 'agentic_traces', + isl: null, + osl: null, + metrics: { + tput_per_gpu: 100, + median_ttft: 0.5, + p75_ttft: 0.9, + p90_ttft: 1.4, + p99_ttft: 6, + mean_ttft: 0.7, + }, + }, + ], + 'agentic-traces', + ); + expect(row.metrics).toEqual({ + tput_per_gpu: 100, + median_ttft: 0.5, + p75_ttft: 0.9, + p90_ttft: 1.4, + }); + }); + it('strips workers and the power audit provenance from the payload-trimmed view', () => { const [row] = toCalculatorBenchmarkRows(rows, '1k/1k'); expect(row).not.toHaveProperty('workers'); diff --git a/packages/app/src/lib/benchmark-api-view.ts b/packages/app/src/lib/benchmark-api-view.ts index c6ef2a8b4..4b1bd19d0 100644 --- a/packages/app/src/lib/benchmark-api-view.ts +++ b/packages/app/src/lib/benchmark-api-view.ts @@ -18,8 +18,10 @@ const CALCULATOR_METRIC_KEYS = new Set([ // GB300 AgentX rows without a server measurement price cached input from the // trace's theoretical ceiling instead. See `pricingCacheHitRate`. 'theoretical_cache_hit_rate', + // `ttft` feeds the First-Token Limits page, which reads this same view and + // caps rows on time to first token at the percentile `intvty` uses. ...['median', 'p75', 'p90'].flatMap((percentile) => - ['intvty', 'itl', 'full_response_itl', 'e2el', 'ttlt'].map( + ['intvty', 'itl', 'full_response_itl', 'e2el', 'ttlt', 'ttft'].map( (metric) => `${percentile}_${metric}`, ), ), diff --git a/packages/app/src/lib/cache-reuse-link.test.ts b/packages/app/src/lib/cache-reuse-link.test.ts new file mode 100644 index 000000000..5ad6d2172 --- /dev/null +++ b/packages/app/src/lib/cache-reuse-link.test.ts @@ -0,0 +1,69 @@ +import { describe, expect, it } from 'vitest'; + +import type { PointMeta } from '@/hooks/api/use-trace-server-metrics'; + +import { cacheReuseHref, pointHardwareKey } from './cache-reuse-link'; + +const point: PointMeta = { + id: 206885, + hardware: 'b200', + framework: 'sglang', + model: 'glm5.2', + precision: 'fp4', + spec_method: 'none', + disagg: false, + is_multinode: false, + conc: 8, + offload_mode: 'on', + kv_offloading: 'dram', + kv_offload_backend: 'hicache', + kv_offload_backend_version: null, + kv_p2p_transfer: null, + router_name: null, + router_version: null, + isl: null, + osl: null, + benchmark_type: 'agentic_traces', + date: '2026-09-17', + run_url: null, + server_gpu_cache_hit_rate: 0.903, + server_cpu_cache_hit_rate: 0.06, +}; + +describe('cacheReuseHref', () => { + it('opens the tab under the reader locale and carries the chart state', () => { + expect(cacheReuseHref('en')).toBe('/cache-reuse'); + expect(cacheReuseHref('zh')).toBe('/zh/cache-reuse'); + }); + + it('pins a point to its own model, scenario, and configuration', () => { + const url = new URL(cacheReuseHref('zh', point), 'https://inferencex.test'); + expect(url.pathname).toBe('/zh/cache-reuse'); + expect(url.searchParams.get('g_model')).toBe('GLM-5.2'); + expect(url.searchParams.get('i_seq')).toBe('agentic-traces'); + expect(url.searchParams.get('i_prec')).toBe('fp4'); + expect(url.searchParams.get('c_cfg')).toBe('b200_sglang'); + }); + + it('leaves unknown model buckets and precisions to the store', () => { + const url = new URL( + cacheReuseHref('en', { ...point, model: 'mystery', precision: 'fp6' }), + 'https://x.test', + ); + expect(url.searchParams.has('g_model')).toBe(false); + expect(url.searchParams.has('i_prec')).toBe(false); + expect(url.searchParams.get('c_cfg')).toBe('b200_sglang'); + }); +}); + +describe('pointHardwareKey', () => { + it('resolves a disaggregated point to the same series key the chart uses', () => { + expect( + pointHardwareKey({ ...point, hardware: 'gb200', framework: 'dynamo-vllm', disagg: true }), + ).toBe('gb200_dynamo-vllm'); + }); + + it('never folds the speculative method into an agentic key', () => { + expect(pointHardwareKey({ ...point, spec_method: 'mtp' })).toBe('b200_sglang'); + }); +}); diff --git a/packages/app/src/lib/cache-reuse-link.ts b/packages/app/src/lib/cache-reuse-link.ts new file mode 100644 index 000000000..2520071d8 --- /dev/null +++ b/packages/app/src/lib/cache-reuse-link.ts @@ -0,0 +1,45 @@ +import { DB_MODEL_TO_DISPLAY } from '@semianalysisai/inferencex-constants'; + +import type { PointMeta } from '@/hooks/api/use-trace-server-metrics'; +import type { AggDataEntry } from '@/components/inference/types'; +import { getHardwareKey } from '@/lib/chart-utils'; +import { PRECISION_OPTIONS, Sequence } from '@/lib/data-mappings'; +import { localePath, type Locale } from '@/lib/i18n'; +import { withChartState } from '@/lib/url-state'; + +/** + * In-app href for the Prefix Cache Reuse tab from an agentic chart or point. + * + * From the chart, the current filters ride along through `withChartState` so + * the tab opens on the same model, scenario, and precisions. From a point + * detail page the reader may have landed cold (no in-memory chart state), so + * the point's own model, precision, and configuration are written explicitly; + * they win over whatever the store still holds because `URLSearchParams.set` + * replaces. The precision must be pinned to one value: with several selected, + * the tab keys its groups as `hwKey__precision` and a bare `c_cfg` would miss. + */ +export function cacheReuseHref(locale: Locale, point?: PointMeta): string { + const href = withChartState(localePath('/cache-reuse', locale)); + if (!point) return href; + const [path, search = ''] = href.split('?'); + const params = new URLSearchParams(search); + const model = DB_MODEL_TO_DISPLAY[point.model]; + if (model) params.set('g_model', model); + params.set('i_seq', Sequence.AgenticTraces); + if ((PRECISION_OPTIONS as readonly string[]).includes(point.precision)) { + params.set('i_prec', point.precision); + } + params.set('c_cfg', pointHardwareKey(point)); + return `${path}?${params.toString()}`; +} + +/** The chart-series key of a point, as the calculator groups it. */ +export function pointHardwareKey(point: PointMeta): string { + return getHardwareKey({ + hw: point.hardware, + framework: point.framework, + disagg: point.disagg, + benchmark_type: point.benchmark_type, + spec_decoding: point.spec_method, + } as unknown as AggDataEntry); +} diff --git a/packages/app/src/lib/chart-utils.test.ts b/packages/app/src/lib/chart-utils.test.ts index ecac454c6..70c43f710 100644 --- a/packages/app/src/lib/chart-utils.test.ts +++ b/packages/app/src/lib/chart-utils.test.ts @@ -889,6 +889,19 @@ describe('buildDerivedChartFields', () => { expect(historicalFields.measuredJPerTotalToken).toBeUndefined(); expect(Object.keys(historicalFields)).toEqual(['tpPerGpu']); }); + + it('emits only the requested power-boundary fields and skips unavailable ones', () => { + // No output throughput and no telemetry: only the spec watts can be derived. + const historicalFields = buildDerivedChartFields(entry(), 'h100', [ + 'tpPerGpu', + 'gpuProvisionedWatts', + 'gpuProvisionedJPerOutputToken', + 'utilityModeledWatts', + ]); + + expect(Object.keys(historicalFields)).toEqual(['tpPerGpu', 'gpuProvisionedWatts']); + expect(historicalFields.gpuProvisionedWatts).toEqual({ y: 700, roof: false }); + }); }); // =========================================================================== @@ -985,6 +998,143 @@ describe('createChartDataPoint energy fields', () => { }); }); +// =========================================================================== +// createChartDataPoint — power-boundary fields (B2 GPU provisioned, B3 utility +// provisioned, B4 utility modeled). Mock specs: tdp 700 W, power 700 "kW". +// =========================================================================== +const boundaryPoint = (e: AggDataEntry) => + createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); + +describe('createChartDataPoint power-boundary fields', () => { + // Eight measured GPUs (2 prefill + 6 decode) on two partially filled chassis: + // the model evaluates 16 GPUs, so only deploymentFacilityWatts ÷ gpuCount yields 877.5 W. + // Every other numerator/denominator pairing gives a different number. + const supportedModel = { + status: 'supported' as const, + hardware: 'h100', + modelRevision: 'test', + modelPath: 'test', + gpuCount: 8, + chassisCount: 2, + modeledGpuCount: 16, + measuredGpuWattsPerGpu: 500, + chassisAcWatts: 10_400, + chassisAcWattsPerGpu: 650, + facilityWatts: 13_520, + deploymentAcWatts: 5400, + deploymentFacilityWatts: 7020, + pue: 1.3, + telemetryBasis: 'validated-v2' as const, + topologyBasis: 'worker-hosts' as const, + chassisBasis: 'extrapolated' as const, + }; + const validated = { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 500, + joules_per_output_token: 10, + modeledSystemPower: supportedModel, + }; + it('emits all six boundary fields for a validated official row', () => { + const p = boundaryPoint( + entry({ output_tput_per_gpu: 400, benchmark_type: 'single_turn', ...validated }), + ); + expect(p.gpuProvisionedWatts).toEqual({ y: 700, roof: false }); + expect(p.gpuProvisionedJPerOutputToken?.y).toBeCloseTo(700 / 400, 10); + expect(p.utilityProvisionedWatts).toEqual({ y: 700_000, roof: false }); + expect(p.utilityProvisionedJPerOutputToken?.y).toBeCloseTo(700_000 / 400, 10); + // B4 W = deployment facility ÷ measured GPUs; J scales B1 by B4 W ÷ B1 W. + expect(p.utilityModeledWatts).toEqual({ y: 877.5, roof: false }); + expect(p.utilityModeledJPerOutputToken?.y).toBeCloseTo((10 * 877.5) / 500, 10); + // Existing measured (B1) fields are untouched by the new boundaries. + expect(p.measuredAvgPower).toEqual({ y: 500, roof: false }); + expect(p.measuredJPerOutputToken).toEqual({ y: 10, roof: false }); + }); + + // The ?unofficialrun= overlay seam (transformBenchmarkRows(rows, 'median', + // 'external') vs the official derivation) is asserted in power-basis.test.ts, + // where real BenchmarkRow fixtures and HW_REGISTRY specs are available. + + it('omits provisioned energies when output throughput is missing, keeps spec watts and B1-scaled modeled energy', () => { + const p = boundaryPoint( + entry({ output_tput_per_gpu: 0, benchmark_type: 'single_turn', ...validated }), + ); + expect(p.gpuProvisionedWatts?.y).toBe(700); + expect(p.utilityProvisionedWatts?.y).toBe(700_000); + expect(p.utilityModeledWatts?.y).toBe(877.5); + expect(p.gpuProvisionedJPerOutputToken).toBeUndefined(); + expect(p.utilityProvisionedJPerOutputToken).toBeUndefined(); + expect('gpuProvisionedJPerOutputToken' in p).toBe(false); + expect('utilityProvisionedJPerOutputToken' in p).toBe(false); + // Modeled energy inherits B1's token denominator, so it survives. + expect(p.utilityModeledJPerOutputToken?.y).toBeCloseTo(17.55, 10); + }); + + it('normalizes fixed-sequence disaggregated energy by all GPUs while jOutput stays per decode GPU', () => { + const p = boundaryPoint( + entry({ + output_tput_per_gpu: 400, + disagg: true, + benchmark_type: 'single_turn', + num_prefill_gpu: 4, + num_decode_gpu: 4, + }), + ); + expect(p.gpuProvisionedJPerOutputToken?.y).toBeCloseTo((700 * 8) / (400 * 4), 10); + expect(p.utilityProvisionedJPerOutputToken?.y).toBeCloseTo((700_000 * 8) / (400 * 4), 10); + expect(p.jOutput?.y).toBeCloseTo(700_000 / 400, 10); + expect(p.utilityProvisionedJPerOutputToken?.y).toBeCloseTo(2 * p.jOutput!.y, 10); + + const agentic = boundaryPoint( + entry({ + output_tput_per_gpu: 400, + disagg: true, + benchmark_type: 'agentic_traces', + num_prefill_gpu: 4, + num_decode_gpu: 4, + }), + ); + expect(agentic.gpuProvisionedWatts?.y).toBe(700); + expect(agentic.gpuProvisionedJPerOutputToken).toBeUndefined(); + expect(agentic.utilityProvisionedJPerOutputToken).toBeUndefined(); + }); + + it('omits the modeled boundary without a supported model or without B1 watts', () => { + const unsupported = boundaryPoint( + entry({ + output_tput_per_gpu: 400, + ...validated, + modeledSystemPower: { status: 'unsupported', reason: 'hardware', modelRevision: 'test' }, + }), + ); + expect(unsupported.utilityModeledWatts).toBeUndefined(); + expect(unsupported.utilityModeledJPerOutputToken).toBeUndefined(); + expect(unsupported.utilityProvisionedJPerOutputToken).toBeDefined(); + + // Admission belongs to the model, not to the schema marker: a validated + // unversioned single-node row the model accepted renders B4 alongside B1. + const legacy = boundaryPoint( + entry({ output_tput_per_gpu: 400, ...validated, power_metric_schema_version: undefined }), + ); + expect(legacy.measuredAvgPower?.y).toBe(500); + expect(legacy.modeledChassisPowerPerGpu?.y).toBe(650); + expect(legacy.utilityModeledWatts?.y).toBe(877.5); + expect(legacy.utilityModeledJPerOutputToken?.y).toBeCloseTo(17.55, 10); + + // B4 is B1 scaled to the meter, so it needs B1's watts even with a model. + const noB1 = boundaryPoint( + entry({ output_tput_per_gpu: 400, ...validated, avg_power_w: undefined }), + ); + expect(noB1.measuredAvgPower).toBeUndefined(); + expect(noB1.utilityModeledWatts).toBeUndefined(); + expect(noB1.utilityModeledJPerOutputToken).toBeUndefined(); + + const noTelemetry = boundaryPoint(entry({ output_tput_per_gpu: 400 })); + expect(noTelemetry.utilityModeledWatts).toBeUndefined(); + expect(noTelemetry.gpuProvisionedJPerOutputToken?.y).toBeCloseTo(1.75, 10); + }); +}); + // =========================================================================== // createChartDataPoint — measured power / energy fields (from runner telemetry) // =========================================================================== @@ -1205,6 +1355,32 @@ describe('createChartDataPoint per-stage measured power fields', () => { expect(point.measuredDecodeJPerOutputToken).toBeDefined(); expect(point.measuredDecodeJPerOutputToken!.y).toBe(0); }); + + it('carries the prefill pool energy onto the output-token axis for validated disaggregated rows', () => { + const e = entry({ + disagg: true, + power_valid: 1, + power_metric_schema_version: 2, + joules_per_input_token: 0.3, + joules_per_output_token: 2.4, + prefill_joules_per_input_token: 0.1, + decode_joules_per_output_token: 1.6, + }); + const point = createChartDataPoint('2025-01-01', e, 'median_e2el', 'tput_per_gpu', 'h100'); + // 0.1 J/in × (2.4 ÷ 0.3 = 8 input tokens per output token) = 0.8 J/out, + // which with the decode pool's 1.6 J/out reconstructs the 2.4 J/out total. + expect(point.reconstructedPrefillJPerOutputToken!.y).toBeCloseTo(0.8, 12); + expect(point.reconstructedPrefillJPerOutputToken!.roof).toBe(false); + // Aggregate rows have no pools to split. + const aggregate = createChartDataPoint( + '2025-01-01', + entry({ ...e, disagg: false }), + 'median_e2el', + 'tput_per_gpu', + 'h100', + ); + expect(aggregate.reconstructedPrefillJPerOutputToken).toBeUndefined(); + }); }); // =========================================================================== diff --git a/packages/app/src/lib/chart-utils.ts b/packages/app/src/lib/chart-utils.ts index e94032342..78b62b2d8 100644 --- a/packages/app/src/lib/chart-utils.ts +++ b/packages/app/src/lib/chart-utils.ts @@ -11,6 +11,7 @@ import type { AggDataEntry, ChartDefinition, InferenceData, + PowerBasisFieldKey, YAxisMetricKey, } from '@/components/inference/types'; import { @@ -20,6 +21,8 @@ import { import { DEFAULT_TCO_BASIS, getGpuSpecs, isKnownGpu, type TcoBasis } from '@/lib/constants'; import { getVendor, type Vendor } from '@/lib/dynamic-colors'; import type { Locale } from '@/lib/i18n'; +import { buildPowerBasisChartFields, type PowerBasisChartFields } from '@/lib/power-basis'; +import { reconstructedRoleEnergy } from '@/components/inference/utils/role-energy'; // --------------------------------------------------------------------------- // High-contrast color generation (iwanthue — k-means in CIELab) @@ -289,7 +292,13 @@ export function buildAvailabilityHwKey( return hwKey; } -export type DerivedMetricKey = BenchmarkMetricKey; +// Power-boundary fields are derived here before the registry exposes them as +// axes; the union collapses once METRIC_REGISTRY carries the same keys. The +// reconstructed prefill energy is a comparison-only series (never an axis). +export type DerivedMetricKey = + | BenchmarkMetricKey + | PowerBasisFieldKey + | 'reconstructedPrefillJPerOutputToken'; export type DerivedChartFields = Pick; const chartMetric = (y: number): { y: number; roof: boolean } => ({ y, roof: false }); @@ -411,6 +420,11 @@ export function buildDerivedChartFields( hardwarePower && tputPerGpu ? (hardwarePower * 1000) / tputPerGpu : 0, ); } + // jOutput keeps the historical per-GPU normalization: for disaggregated rows + // output_tput_per_gpu is per decode GPU, so this is all-in W of one decode GPU + // per output token and ignores the prefill pool. The power-boundary field + // utilityProvisionedJPerOutputToken uses the same all-in W but counts every + // allocated GPU, so the two differ on disaggregated rows by (P + D) / D. if (hardwarePower > 0 && wants('jOutput') && outputTputPerGpu) { fields.jOutput = chartMetric(hardwarePower ? (hardwarePower * 1000) / outputTputPerGpu : 0); } @@ -426,6 +440,14 @@ export function buildDerivedChartFields( if (wants(key)) fields[key] = value; } + const powerBasis = buildPowerBasisChartFields(entry, specs); + for (const [key, value] of Object.entries(powerBasis) as [ + keyof PowerBasisChartFields, + { y: number; roof: boolean }, + ][]) { + if (wants(key)) fields[key] = value; + } + if (wants('modeledChassisPowerPerGpu') && entry.modeledSystemPower?.status === 'supported') { fields.modeledChassisPowerPerGpu = chartMetric(entry.modeledSystemPower.chassisAcWattsPerGpu); } @@ -521,6 +543,8 @@ type MeasuredPowerChartFields = Partial< | 'measuredJPerSuccessfulQuery' | 'measuredWhPerSuccessfulQuery' | 'measuredPowerPercentTdp' + | 'measuredPowerTimeline' + | 'reconstructedPrefillJPerOutputToken' > >; @@ -530,8 +554,13 @@ function buildMeasuredPowerChartFields( tdpWatts: number, ): MeasuredPowerChartFields { return { + // The timeline axis aliases the validated average: the point set (and + // its table row) is the same, only the chart body changes. ...(typeof entry.avg_power_w === 'number' - ? { measuredAvgPower: chartMetric(entry.avg_power_w) } + ? { + measuredAvgPower: chartMetric(entry.avg_power_w), + measuredPowerTimeline: chartMetric(entry.avg_power_w), + } : {}), ...(typeof entry.p75_power_w === 'number' && Number.isFinite(entry.p75_power_w) ? { measuredP75Power: chartMetric(entry.p75_power_w) } @@ -562,6 +591,14 @@ function buildMeasuredPowerChartFields( ...(typeof entry.decode_joules_per_output_token === 'number' ? { measuredDecodeJPerOutputToken: chartMetric(entry.decode_joules_per_output_token) } : {}), + // Prefill energy on the output-token axis, so the roles comparison can + // stack it against the decode pool (PowerX Figure 7). + ...(() => { + const roleEnergy = reconstructedRoleEnergy(entry); + return roleEnergy + ? { reconstructedPrefillJPerOutputToken: chartMetric(roleEnergy.prefill) } + : {}; + })(), ...(typeof entry.joules_per_successful_query === 'number' ? { measuredJPerSuccessfulQuery: chartMetric(entry.joules_per_successful_query), diff --git a/packages/app/src/lib/compare-slug.test.ts b/packages/app/src/lib/compare-slug.test.ts index c5744203b..f3ab5a02a 100644 --- a/packages/app/src/lib/compare-slug.test.ts +++ b/packages/app/src/lib/compare-slug.test.ts @@ -274,6 +274,8 @@ describe('compareModelSeoName', () => { 'minimax-m3': 'MiniMax M3', 'minimax-m27': 'MiniMax M2.7', 'qwen-3-8-flash-next': 'Qwen3.8-Flash-Next', + 'qwen-3-8-27b': 'Qwen3.8-27B', + 'qwen-3-8-27b-eager': 'Qwen3.8-27B Eager', 'qwen-3-5': 'Qwen3.5', 'gptoss-120b': 'gpt-oss-120b', 'llama-3-3-70b': 'Llama 3.3 70B', diff --git a/packages/app/src/lib/compare-slug.ts b/packages/app/src/lib/compare-slug.ts index 59930956c..3619198c4 100644 --- a/packages/app/src/lib/compare-slug.ts +++ b/packages/app/src/lib/compare-slug.ts @@ -141,6 +141,22 @@ export const COMPARE_MODEL_SLUGS: CompareModelSlug[] = [ label: 'Qwen 3.8 Flash Next 176B-A6B', seoName: 'Qwen3.8-Flash-Next', }, + { + slug: 'qwen-3-8-27b', + displayName: 'Qwen3.8-27B', + dbKeys: ['qwen3.827b'], + // Dense 27B: total and active parameter counts coincide. + label: 'Qwen 3.8 27B', + seoName: 'Qwen3.8-27B', + }, + { + slug: 'qwen-3-8-27b-eager', + displayName: 'Qwen3.8-27B-Eager', + dbKeys: ['qwen3.827beager'], + // Same checkpoint with CUDA graphs disabled; kept as its own bucket. + label: 'Qwen 3.8 27B (eager)', + seoName: 'Qwen3.8-27B Eager', + }, { slug: 'qwen-3-5', displayName: 'Qwen-3.5-397B-A17B', diff --git a/packages/app/src/lib/compare-ssr.ts b/packages/app/src/lib/compare-ssr.ts index a1f8a346f..9365caa6f 100644 --- a/packages/app/src/lib/compare-ssr.ts +++ b/packages/app/src/lib/compare-ssr.ts @@ -49,6 +49,8 @@ export const KNOWN_MODELS = new Set([ 'gpt-oss-120b', 'Qwen-3.5-397B-A17B', 'Qwen3.8-Flash-Next', + 'Qwen3.8-27B', + 'Qwen3.8-27B-Eager', 'Kimi-K2.5', 'Kimi-K3', 'MiniMax-M2.5', diff --git a/packages/app/src/lib/csv-export-helpers.test.ts b/packages/app/src/lib/csv-export-helpers.test.ts index b850da5b3..467115ae6 100644 --- a/packages/app/src/lib/csv-export-helpers.test.ts +++ b/packages/app/src/lib/csv-export-helpers.test.ts @@ -749,3 +749,33 @@ describe('historicalTrendToCsv (mirrors HistoricalTrendsDisplay export)', () => expect(rows[1][headers.indexOf('Date')]).toBe('2025-01-15'); }); }); + +describe('inferenceChartToCsv power comparison', () => { + it('names each row’s boundary or role only while comparison clones are present', () => { + const base = makePoint({ + hwKey: 'b200_sglang', + measuredAvgPower: { y: 600, roof: false }, + }); + const clone = { + ...base, + y: 1000, + powerVariant: { kind: 'basis', id: 'gpu-provisioned' }, + } as InferenceData; + const plain = inferenceChartToCsv([base], 'dsv4', '8k/1k', [], { + yHeader: 'Measured Power per Chip (W)', + yPath: 'measuredAvgPower.y', + xHeader: 'Interactivity (tok/s/user)', + }); + expect(plain.headers).not.toContain('Power Series'); + + const { headers, rows } = inferenceChartToCsv([base, clone], 'dsv4', '8k/1k', [], { + yHeader: 'Measured Power per Chip (W)', + yPath: 'measuredAvgPower.y', + xHeader: 'Interactivity (tok/s/user)', + }); + const column = headers.indexOf('Power Series'); + expect(column).toBeGreaterThan(-1); + expect(rows.map((row) => row[column])).toEqual(['GPU measured', 'GPU provisioned (TDP)']); + expect(rows.every((row) => row.length === headers.length)).toBe(true); + }); +}); diff --git a/packages/app/src/lib/csv-export-helpers.ts b/packages/app/src/lib/csv-export-helpers.ts index 576569af8..f4803a150 100644 --- a/packages/app/src/lib/csv-export-helpers.ts +++ b/packages/app/src/lib/csv-export-helpers.ts @@ -9,6 +9,7 @@ import { METRIC_REGISTRY } from '@/components/inference/metric-registry'; import type { InferenceData, TrendDataPoint } from '@/components/inference/types'; +import { inferPowerCompare, powerSeriesLabel } from '@/components/inference/utils/power-compare'; import { chipCounts } from '@/lib/chip-counts'; import type { SubmissionVolumeRow } from '@/lib/submissions-types'; @@ -57,6 +58,12 @@ export function inferenceChartToCsv( const islOsl = sequenceToIslOsl(sequence); const showModeledPower = displayedMetrics?.yPath === METRIC_REGISTRY.modeledChassisPowerPerGpu.field; + // A power comparison (`i_pcompare`) appends boundary / role clones of the + // plotted points; name each row's series so the export stays unambiguous. + const allPoints = [...data, ...overlayData]; + const powerCompare = inferPowerCompare(allPoints); + const showPowerSeries = powerCompare !== 'none'; + const plottedMetric = displayedMetrics ? `y_${displayedMetrics.yPath.split('.')[0]}` : ''; const headers = [ 'Model', 'ISL', @@ -111,6 +118,7 @@ export function inferenceChartToCsv( 'Physical Chips', 'DP', ...(showModeledPower ? ['Configured Chip Count'] : []), + ...(showPowerSeries ? ['Power Series'] : []), ]; const displayedColumns = displayedMetrics @@ -128,7 +136,7 @@ export function inferenceChartToCsv( : []; headers.splice(10, 0, ...displayedColumns.map((column) => column.header)); - const rows = [...data, ...overlayData] + const rows = allPoints .filter((d) => !d.hidden) .map((d) => { const chips = chipCounts(d, showModeledPower); @@ -177,6 +185,7 @@ export function inferenceChartToCsv( chips.physical, d.dp ?? '', ...(showModeledPower ? [chips.configured] : []), + ...(showPowerSeries ? [powerSeriesLabel(d, plottedMetric, powerCompare, 'en')] : []), ]; row.splice(10, 0, ...displayedColumns.map((column) => column.value(d))); return row; diff --git a/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts b/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts index d50fe1573..428fcbf8f 100644 --- a/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts +++ b/packages/app/src/lib/d3-chart/layers/perf-ruler.test.ts @@ -16,10 +16,12 @@ import { isPerfRulerCurveVisible, movePerfRulerIsoX, nextPerfRulerState, + parsePerfRulers, pathXExtent, perfRulerCurveSet, prunePerfRulers, renderPerfRulers, + serializePerfRulers, type PerfRulerEndInput, type PerfRulerGeometry, type PerfRulerLabelLayoutOptions, @@ -889,6 +891,132 @@ describe('perfRulerCurveSet', () => { }); }); +// ── serializePerfRulers / parsePerfRulers (share links) ───────────── + +describe('serializePerfRulers / parsePerfRulers', () => { + const OFFICIAL_A = 'roofline-b200_trt_fp8'; + const OFFICIAL_B = 'roofline-mi355x_sglang_fp4'; + // Overlay curves carry the unofficial run index; power-envelope curves are + // split per date with the encoded date appended (`%2F` from a slash). + const OVERLAY = 'overlay-roofline-h100_vllm_fp8_run1__2026-09%2F11'; + + it('round-trips completed rulers, including overlay and date-scoped curve ids', () => { + const state = complete( + complete(EMPTY_PERF_RULER_STATE, OFFICIAL_A, OFFICIAL_B, 41.5), + OFFICIAL_A, + OVERLAY, + 120, + ); + const encoded = serializePerfRulers(state); + expect(encoded).toBe(`41.5|${OFFICIAL_A}|${OFFICIAL_B};120|${OFFICIAL_A}|${OVERLAY}`); + const parsed = parsePerfRulers(encoded); + expect(parsed.rulers).toEqual([ + { id: 1, curveA: OFFICIAL_A, curveB: OFFICIAL_B, isoX: 41.5 }, + { id: 2, curveA: OFFICIAL_A, curveB: OVERLAY, isoX: 120 }, + ]); + expect(parsed.draft).toBeNull(); + expect(parsed.nextId).toBe(3); + }); + + it('survives URLSearchParams encoding, which the share link applies', () => { + const state = complete(EMPTY_PERF_RULER_STATE, OFFICIAL_A, OVERLAY, 33); + const search = new URLSearchParams({ i_rulers: serializePerfRulers(state) }); + const decoded = new URLSearchParams(search.toString()).get('i_rulers'); + expect(parsePerfRulers(decoded).rulers).toEqual([ + { id: 1, curveA: OFFICIAL_A, curveB: OVERLAY, isoX: 33 }, + ]); + }); + + it('rounds the iso-x to four significant digits and never emits the draft', () => { + let state = complete(EMPTY_PERF_RULER_STATE, OFFICIAL_A, OFFICIAL_B, 123.456789); + state = nextPerfRulerState(state, { curve: OFFICIAL_B, isoX: 9 }); + expect(state.draft).not.toBeNull(); + expect(serializePerfRulers(state)).toBe(`123.5|${OFFICIAL_A}|${OFFICIAL_B}`); + expect(serializePerfRulers(complete(EMPTY_PERF_RULER_STATE, 'a', 'b', 0.00012346))).toBe( + '0.0001235|a|b', + ); + expect(serializePerfRulers(complete(EMPTY_PERF_RULER_STATE, 'a', 'b', 98765))).toBe( + '98770|a|b', + ); + }); + + it('serializes the empty state to the empty string so the param strips as default', () => { + expect(serializePerfRulers(EMPTY_PERF_RULER_STATE)).toBe(''); + const drafted = nextPerfRulerState(EMPTY_PERF_RULER_STATE, { curve: 'a', isoX: 1 }); + expect(serializePerfRulers(drafted)).toBe(''); + }); + + it('returns the shared empty state for missing or entirely malformed input', () => { + expect(parsePerfRulers('')).toBe(EMPTY_PERF_RULER_STATE); + expect(parsePerfRulers(null)).toBe(EMPTY_PERF_RULER_STATE); + expect(parsePerfRulers(undefined)).toBe(EMPTY_PERF_RULER_STATE); + expect(parsePerfRulers('garbage')).toBe(EMPTY_PERF_RULER_STATE); + expect(parsePerfRulers(';;|')).toBe(EMPTY_PERF_RULER_STATE); + }); + + it('drops malformed entries and keeps the valid ones', () => { + const encoded = [ + `NaN|${OFFICIAL_A}|${OFFICIAL_B}`, // non-numeric iso-x + `|${OFFICIAL_A}|${OFFICIAL_B}`, // missing iso-x + `12|${OFFICIAL_A}`, // missing curve B + `12||${OFFICIAL_B}`, // empty curve A + `12|${OFFICIAL_A}|${OFFICIAL_A}`, // a curve cannot be measured against itself + `12|${OFFICIAL_A}|${OFFICIAL_B}|extra`, // too many fields + `Infinity|${OFFICIAL_A}|${OFFICIAL_B}`, + `7.5|${OFFICIAL_A}|${OFFICIAL_B}`, + ].join(';'); + expect(parsePerfRulers(encoded).rulers).toEqual([ + { id: 1, curveA: OFFICIAL_A, curveB: OFFICIAL_B, isoX: 7.5 }, + ]); + }); + + it('only accepts roofline identity classes as curve ids', () => { + const encoded = [ + // The shared marker classes sit on EVERY roofline path: a ruler between + // them would land on whichever two paths come first in the DOM. + `10|roofline-path|overlay-roofline-path`, + `10|${OFFICIAL_A}|roofline-path`, + // Other nodes inside the zoom group are not curves. + `10|perf-ruler-hit|${OFFICIAL_A}`, + `10|dot-group|zoom-group`, + `10|${OFFICIAL_A}|roofline-`, // identity class needs a series + `10|${OFFICIAL_A}|roofline-b200 fp8`, // no whitespace in a class token + `10|${OFFICIAL_A}|${OVERLAY}`, + ].join(';'); + expect(parsePerfRulers(encoded).rulers).toEqual([ + { id: 1, curveA: OFFICIAL_A, curveB: OVERLAY, isoX: 10 }, + ]); + }); + + it('caps at MAX_PERF_RULERS keeping the newest entries, like the click reducer', () => { + const encoded = Array.from( + { length: MAX_PERF_RULERS + 3 }, + (_, index) => `${index + 1}|roofline-a${index}|roofline-b${index}`, + ).join(';'); + const parsed = parsePerfRulers(encoded); + expect(parsed.rulers).toHaveLength(MAX_PERF_RULERS); + expect(parsed.rulers[0]).toEqual({ + id: 1, + curveA: 'roofline-a3', + curveB: 'roofline-b3', + isoX: 4, + }); + expect(parsed.rulers.at(-1)).toEqual({ + id: MAX_PERF_RULERS, + curveA: `roofline-a${MAX_PERF_RULERS + 2}`, + curveB: `roofline-b${MAX_PERF_RULERS + 2}`, + isoX: MAX_PERF_RULERS + 3, + }); + expect(parsed.nextId).toBe(MAX_PERF_RULERS + 1); + }); + + it('hands out ids a later click never reuses', () => { + const restored = parsePerfRulers(`10|roofline-a|roofline-b;20|roofline-c|roofline-d`); + const placed = complete(restored, 'e', 'f', 30); + expect(placed.rulers.map((ruler) => ruler.id)).toEqual([1, 2, 3]); + }); +}); + // ── pathXExtent ───────────────────────────────────────────── describe('pathXExtent', () => { diff --git a/packages/app/src/lib/d3-chart/layers/perf-ruler.ts b/packages/app/src/lib/d3-chart/layers/perf-ruler.ts index c44143529..ef13b2a1d 100644 --- a/packages/app/src/lib/d3-chart/layers/perf-ruler.ts +++ b/packages/app/src/lib/d3-chart/layers/perf-ruler.ts @@ -323,6 +323,72 @@ export function prunePerfRulers( return { ...prev, rulers, draft }; } +const PERF_RULER_URL_RULER_SEPARATOR = ';'; +const PERF_RULER_URL_FIELD_SEPARATOR = '|'; +/** + * Shape of a curve id the link may reference: one roofline path's identity + * class (`roofline-` / `overlay-roofline-`), never the shared + * `roofline-path` / `overlay-roofline-path` marker classes or any other node + * inside the zoom group — those match many paths, so a hand-edited link + * would draw a ruler between whichever two come first in DOM order. + */ +const PERF_RULER_CURVE_ID = /^(?:overlay-)?roofline-(?!path$)[\w%.-]+$/u; + +/** + * Share-link encoding of the COMPLETED rulers (`i_rulers`). One ruler per + * `;`, fields joined by `|`: `isoX|curveA|curveB`. Curve ids are the rendered + * roofline path identity classes (`roofline-_`, + * `overlay-roofline-__run`, optionally `__`), whose alphabet is `[A-Za-z0-9_%.-]`, so neither separator can + * appear inside one; `URLSearchParams` percent-encodes both on the wire. + * The iso-x is rounded to four significant digits to keep links short — a + * 0.05% shift on the x metric is far below the ruler's visual resolution. + * The draft is never serialized: it is an unfinished click, not a + * measurement. Empty state serializes to '' so the param strips as default. + */ +export function serializePerfRulers(state: PerfRulerState): string { + return state.rulers + .map((ruler) => + [Number(ruler.isoX.toPrecision(4)), ruler.curveA, ruler.curveB].join( + PERF_RULER_URL_FIELD_SEPARATOR, + ), + ) + .join(PERF_RULER_URL_RULER_SEPARATOR); +} + +/** + * Inverse of {@link serializePerfRulers}. Malformed entries (wrong field + * count, non-numeric iso-x, identical curve ids, or ids that are not + * roofline identity classes) are dropped silently — a hand-edited or + * truncated link degrades to fewer rulers, never to an error. The list is capped at + * {@link MAX_PERF_RULERS} keeping the NEWEST (last-serialized) entries, the + * same end the click reducer drops from. Ids are reassigned 1..n with + * `nextId = n + 1`, so parsed rulers are valid D3 join keys and a ruler + * placed afterwards never collides. Returns {@link EMPTY_PERF_RULER_STATE} + * (same reference) for '', null, or an all-malformed value. + */ +export function parsePerfRulers(raw: string | null | undefined): PerfRulerState { + if (!raw) return EMPTY_PERF_RULER_STATE; + const parsed: Omit[] = []; + for (const entry of raw.split(PERF_RULER_URL_RULER_SEPARATOR)) { + const fields = entry.split(PERF_RULER_URL_FIELD_SEPARATOR); + if (fields.length !== 3) continue; + const [isoXField, curveA, curveB] = fields; + if (isoXField.trim() === '' || curveA === curveB) continue; + if (!PERF_RULER_CURVE_ID.test(curveA) || !PERF_RULER_CURVE_ID.test(curveB)) continue; + const isoX = Number(isoXField); + if (!Number.isFinite(isoX)) continue; + parsed.push({ curveA, curveB, isoX }); + } + if (parsed.length === 0) return EMPTY_PERF_RULER_STATE; + const kept = parsed.slice(-MAX_PERF_RULERS); + return { + rulers: kept.map((ruler, index) => ({ id: index + 1, ...ruler })), + draft: null, + nextId: kept.length + 1, + }; +} + /** Every curve referenced by any ruler or the draft (hit-halo styling). */ export function perfRulerCurveSet(state: PerfRulerState): Set { const curves = new Set(); diff --git a/packages/app/src/lib/dashboard-routes.ts b/packages/app/src/lib/dashboard-routes.ts index 4934ecde3..131e09a33 100644 --- a/packages/app/src/lib/dashboard-routes.ts +++ b/packages/app/src/lib/dashboard-routes.ts @@ -164,6 +164,26 @@ export const DASHBOARD_ROUTES = [ providers: UNOFFICIAL_ONLY_DASHBOARD_PROVIDERS, shareParamScopes: ['g_', 'i_', 'c_'], }, + { + key: 'first-token', + path: '/first-token', + canonicalPath: '/first-token', + navGroup: 'footer-only', + indexable: true, + localeMirrored: true, + providers: UNOFFICIAL_ONLY_DASHBOARD_PROVIDERS, + shareParamScopes: ['g_', 'i_', 'c_'], + }, + { + key: 'cache-reuse', + path: '/cache-reuse', + canonicalPath: '/cache-reuse', + navGroup: 'footer-only', + indexable: true, + localeMirrored: true, + providers: UNOFFICIAL_ONLY_DASHBOARD_PROVIDERS, + shareParamScopes: ['g_', 'i_', 'c_'], + }, { key: 'reliability', path: '/reliability', diff --git a/packages/app/src/lib/data-mappings.ts b/packages/app/src/lib/data-mappings.ts index 4d87b3c54..a13078c2c 100644 --- a/packages/app/src/lib/data-mappings.ts +++ b/packages/app/src/lib/data-mappings.ts @@ -7,6 +7,8 @@ export enum Model { GptOss = 'gpt-oss-120b', Qwen3_5 = 'Qwen-3.5-397B-A17B', Qwen3_8_Flash_Next = 'Qwen3.8-Flash-Next', + Qwen3_8_27B = 'Qwen3.8-27B', + Qwen3_8_27B_Eager = 'Qwen3.8-27B-Eager', Kimi_K2_5 = 'Kimi-K2.5', Kimi_K3 = 'Kimi-K3', MiniMax_M2_5 = 'MiniMax-M2.5', @@ -231,6 +233,27 @@ const MODEL_CONFIG: Record = { openRouterModelId: 'qwen/qwen3.8-flash', logo: 'qwen-color.svg', }, + // Dense 27B (27B active) on the Qwen3.8 hybrid GDN backbone, served bf16 on + // one GPU with the RadixArk DSpark drafter (single-turn 1k1k, InferenceX#3260). + // Experimental until the first sweeps land on the official pipeline. + [Model.Qwen3_8_27B]: { + label: 'Qwen3.8 27B', + prefix: 'qwen3.827b', + category: 'experimental', + openRouterModelId: 'qwen/qwen3.8-27b', + logo: 'qwen-color.svg', + }, + // The same checkpoint served with CUDA graphs disabled on both the target and + // the DSpark drafter (--enforce-eager, InferenceX#3262). A separate bucket so + // the eager and graph-captured points never overlap on one chart. + [Model.Qwen3_8_27B_Eager]: { + label: 'Qwen3.8 27B (eager)', + prefix: 'qwen3.827beager', + category: 'experimental', + // Same checkpoint, so the same public catalog id. + openRouterModelId: 'qwen/qwen3.8-27b', + logo: 'qwen-color.svg', + }, [Model.GptOss]: { label: 'gpt-oss 120B', prefix: 'gptoss', diff --git a/packages/app/src/lib/github-artifacts.test.ts b/packages/app/src/lib/github-artifacts.test.ts index 4a80ae8a4..450623b48 100644 --- a/packages/app/src/lib/github-artifacts.test.ts +++ b/packages/app/src/lib/github-artifacts.test.ts @@ -8,6 +8,7 @@ import { fetchGithubRunArtifacts, getRunDate, normalizeGithubRunInfo, + readZipEntries, type GithubArtifact, type GithubWorkflowRun, } from './github-artifacts'; @@ -140,3 +141,25 @@ describe('extractZipEntries', () => { expect(parseErrors).toEqual(['bad.json']); }); }); + +describe('readZipEntries', () => { + it('decodes only the entries the predicate selects and skips directories', () => { + const zip = new AdmZip(); + zip.addFile('LOGS/power/', Buffer.alloc(0)); + zip.addFile('LOGS/power/samples.csv', Buffer.from('a,b\n1,2', 'utf8')); + zip.addFile('LOGS/big/results.json', Buffer.alloc(64 * 1024, 0x41)); + zip.addFile('power_validation_x_conc8.json', Buffer.from('{"power_valid":true}', 'utf8')); + + const files = readZipEntries( + zip.toBuffer(), + (name) => name === 'LOGS/power/samples.csv' || name.startsWith('power_validation_'), + ); + + expect([...files.keys()].toSorted()).toEqual([ + 'LOGS/power/samples.csv', + 'power_validation_x_conc8.json', + ]); + expect(files.get('LOGS/power/samples.csv')).toBe('a,b\n1,2'); + expect(files.get('power_validation_x_conc8.json')).toBe('{"power_valid":true}'); + }); +}); diff --git a/packages/app/src/lib/github-artifacts.ts b/packages/app/src/lib/github-artifacts.ts index 3fe278903..97d6ac01d 100644 --- a/packages/app/src/lib/github-artifacts.ts +++ b/packages/app/src/lib/github-artifacts.ts @@ -120,6 +120,24 @@ export function downloadGithubArtifact(url: string, token: string): Promise boolean, +): Map { + const zip = new AdmZip(buffer); + const files = new Map(); + for (const entry of zip.getEntries()) { + if (entry.isDirectory || !predicate(entry.entryName)) continue; + files.set(entry.entryName, entry.getData().toString('utf8')); + } + return files; +} + export function extractZipEntries( buffer: Buffer, extension: string, diff --git a/packages/app/src/lib/glossary-zh.ts b/packages/app/src/lib/glossary-zh.ts index 5c6d59415..015652e8c 100644 --- a/packages/app/src/lib/glossary-zh.ts +++ b/packages/app/src/lib/glossary-zh.ts @@ -146,13 +146,14 @@ const translations: Readonly> = { term: '吞吐量', aliases: ['throughput', 'token 吞吐量', '总吞吐量'], plainEnglish: '吞吐量就是整个系统每秒一共能完成多少工作。', - definition: '吞吐量是推理系统在所有活跃请求上生成 token 的总速率。', + definition: + '吞吐量是系统处理各请求中 token 的速率。输出吞吐量只计生成的 token;总 token 吞吐量则按基准测试规定的统计口径,计入输入和输出 token。', explanation: 'InferenceX 通常使用每芯片每秒 token 数进行归一化,便于比较不同规模的系统。提高批大小或并发往往能摊薄权重读取和计算成本,从而提高总吞吐量,但单个用户收到 token 的速度可能下降。', significance: '最大吞吐量不是完整的性能结论。某个点即使拥有最高 tok/s,也可能因为交互性过低而不适合实时产品;有效比较应在符合业务需求的延迟或交互性目标下进行。', benchmarkContext: - 'InferenceX 将吞吐量与交互性放在完整并发扫描中共同展示,并用 Pareto 前沿剔除两个轴上都更差的运行点。', + 'InferenceX 将吞吐量与交互性放在完整并发扫描中共同展示。Rubin AgentX 文章报告的是包含复用输入在内的总 token 吞吐量,不只是新生成的输出。因此,与只统计输出的结果比较前,必须先核对 token 口径。', measurement: { label: '常用单位', value: 'tok/s/chip' }, }, interactivity: { @@ -165,7 +166,7 @@ const translations: Readonly> = { significance: '不同产品需要不同运行点。语音和交互式编程要求较高 token 速率,离线摘要则可以牺牲交互性换取更高总吞吐量;在交互性不一致时比较硬件很容易得出误导性结论。', benchmarkContext: - 'InferenceX 将 tok/s/user 与吞吐量或成本一起绘制,并在等交互性表格中沿各自 Pareto 前沿插值,以固定用户体验。由于该坐标轴不计入首 token 之前的等待,agentic 图表还提供端到端归一化交互性,用同一单位把 TTFT 一并纳入。', + 'InferenceX 将 tok/s/user 与吞吐量或成本一起绘制。Rubin AgentX 文章用 P90 全响应 token 间延迟的倒数表示 P90 交互性。匹配这一速度不代表 TTFT 或端到端延迟也相同;端到端归一化交互性是另一项指标,会把开始输出前的等待计入。', measurement: { label: '常用单位', value: 'token/秒/用户(tok/s/user)' }, }, latency: { @@ -253,7 +254,7 @@ const translations: Readonly> = { explanation: '不同方案的并发点很少正好落在相同 tok/s/user。等交互性比较会在各自 Pareto 前沿上对共同目标插值,再比较该点的吞吐量、成本或效率。', significance: - '固定用户体验可以避免常见基准错误:某系统只有在让每个请求更慢时才达到更高吞吐量,却被错误地称为更快。', + '匹配流式输出速度,可以避免仅因系统以更慢的 token 速率服务更多请求,就把它称为更快。但这并未固定完整的用户体验:首 token 前的等待和整段响应耗时仍需单独比较。', benchmarkContext: 'InferenceX 文章使用等交互性表格比较硬件、精度和软件;超出实测前沿的值会标记为不可达,而不会向观测区间之外外推。前沿始终建立在吞吐量与交互性之上,每百万 token 成本和每 token 焦耳则由插值得到的吞吐量推导,而不是各自单独做样条:它们都是每芯片常数除以吞吐量,单独插值会破坏两个 knot 之间的这一恒等关系。', }, @@ -297,7 +298,7 @@ const translations: Readonly> = { significance: '芯片峰值 FLOPS 不能单独决定服务经济性;内存、网络、软件成熟度、数值精度和实际利用率都会影响最终比值。', benchmarkContext: - 'InferenceX 在匹配交互性时比较基础设施 perf/$,并明确使用的 TCO 输入。该比值不能跨模型、序列长度、精度或延迟区间直接套用。每百万 token 成本以及总 token、输入 token 和输出 token 购买力轴都采用这套 TCO 经济性口径。', + 'InferenceX 在匹配交互性时比较基础设施 perf/$,并注明 TCO 假设。Rubin 文章约 67 倍的结果限定于 170 TPS、自有成本口径及文中指定的 TRTLLM NVFP4 Dense 配置,并非适用于整个硬件世代的倍率。目标速度、对比引擎以及自有或租赁成本口径变化时,比值也会变化。', }, 'total-cost-of-ownership': { term: '总体拥有成本', @@ -309,7 +310,7 @@ const translations: Readonly> = { significance: 'TCO 比标价更适合跨系统经济性比较,尤其是网络与电力基础设施不同的机架级产品;但它仍是模型,必须连同假设一起阅读。', benchmarkContext: - 'InferenceX 将 SemiAnalysis AI Cloud 的 TCO 输入与实测 tok/s/chip 结合,从而把系统每小时成本与决定这一小时 token 产出的软件实现及工作负载特征分开考察。', + 'InferenceX 将 SemiAnalysis AI Cloud 的 TCO 输入与实测 tok/s/chip 结合。Rubin 文章区分了两种口径:超大规模采购条件下的自有成本,包含硬件、网络、机房、电力和资本成本的摊销;以及客户支付的三年云服务预留价格。两者是可选的成本基础,不能相加;租赁价格还反映服务商的商业条款。', }, 'tokens-per-megawatt': { term: '每兆瓦 token 吞吐量', @@ -321,7 +322,7 @@ const translations: Readonly> = { significance: '电力供应往往是新增 AI 部署的硬约束。每兆瓦生成更多 token 的系统,即使单个加速器功耗更高,也能在相同电力配额下服务更多需求。', benchmarkContext: - '比较 tokens/MW 时必须匹配模型、工作负载、精度与交互性,否则高吞吐低交互点可能看似高效,却无法满足目标用户体验。每 token 能耗表达的是同一份供电预算折算到单位输出上的结果;在遥测可信的前提下,InferenceX 还会给出加速器的实测能耗。', + 'Rubin 文章在相同 P90 交互性下比较每兆瓦市电容量对应的总 tok/s,且只在各引擎实测区间内插值。引用倍率时须保留引擎、精度、缓存和并行配置。配置的市电功率分母属于容量模型,与加速器实测功耗或能耗遥测不同。', measurement: { label: '常用单位', value: '每单位配置市电兆瓦的 token/秒' }, }, prefill: { @@ -368,7 +369,7 @@ const translations: Readonly> = { '前缀缓存会记住重复开头的处理结果,例如相同系统提示词,让模型下次可以跳过这部分工作。', definition: '当前多个请求以相同 token 序列开头时,前缀缓存会复用已有 KV 缓存状态。', explanation: - '重复系统提示词、共享文档或共同对话前缀在缓存仍可用时无需再次预填充。命中缓存可显著减少提示词计算与首 token 时间。', + '重复系统提示词、共享文档或共同对话前缀可以复用缓存状态。智能体会话持续增长时,前几轮的输出会成为后续输入,可复用前缀也随之增加。但实际命中仍要求相关状态尚未被淘汰且能够访问;缓存淘汰或子智能体的新上下文都可能带来新的 prefill。', significance: '具有重复前缀的生产工作负载可能明显快于随机 token 基准;收益取决于命中率、缓存容量、淘汰策略与请求能否路由到持有所需状态的节点。', benchmarkContext: @@ -544,7 +545,7 @@ const translations: Readonly> = { plainEnglish: 'NVLink 是 NVIDIA 芯片之间的高速公路,让多张芯片的协作远快于普通服务器网络。', definition: 'NVLink 是 NVIDIA 用于 scale-up 域内芯片直接数据传输的高带宽加速器互连。', explanation: - 'NVSwitch 系统连接多个 NVLink 端点,使集体通信可覆盖八卡服务器,或在 NVL72 产品中覆盖 72 芯片机架级域;该带宽不同于连接独立系统的 InfiniBand/Ethernet。', + 'NVSwitch 连接多个 NVLink 端点,使集合通信可以覆盖单节点或包含 72 颗芯片的机架级域。互连代际取决于平台:Blackwell NVL72 使用 NVLink 5,Rubin 文章则将 NVLink 6 Switch 列为 Vera Rubin 的组成部分。这种 scale-up 互连与系统之间的网络不同。', significance: '大型 TP,尤其是 Wide EP,会在每个生成 token 上交换数据。把通信留在 NVLink 上,可让机架级方案显著快于通过 scale-out 连接的相似芯片数量。', benchmarkContext: @@ -1125,13 +1126,13 @@ const translations: Readonly> = { }, nvl72: { term: 'NVL72', - aliases: ['NVL72', 'GB200 NVL72', 'GB300 NVL72', '机架级系统'], + aliases: ['NVL72', 'GB200 NVL72', 'GB300 NVL72', 'Vera Rubin NVL72', '机架级系统'], plainEnglish: 'NVL72 是一个机架,其中 72 个加速器共享同一张高速网络,因而更像一台大机器而不是一个集群。', definition: 'NVL72 是 NVIDIA 的机架级系统,把 72 个加速器放进同一个 NVLink scale-up 域,而不是分散在多个八芯片节点中。', explanation: - '仪表板的规格数据记录为 NVLink 5.0、每颗芯片 900 GB/s 单向带宽、scale-up world size 为 72,并通过 NVSwitch 交换。常规节点把同样的带宽限定在八颗芯片之间,超出后就要退回更慢的 scale-out 网络,因此差别不在于原始速度,而在于换用另一种网络之前能触及多少颗芯片。', + 'NVL72 描述的是 NVLink 域的规模,并不限定某一代芯片或固定带宽。GB200 和 GB300 使用 Blackwell 世代硬件及 NVLink 5;Vera Rubin 则组合 Rubin GPU、Vera CPU 和 NVLink 6 Switch。比较时应查看具体平台,不能把 Blackwell 规格套用到所有 NVL72 机架。', significance: '开销以集合通信为主的技术,在大规模 scale-up 域中经济性会发生变化。宽专家并行把专家分散到许多芯片上,每个 token 都要付出 all-to-all 流量;在 scale-up 带宽下这可以承受,在 scale-out 网络上往往不行。', benchmarkContext: @@ -1494,7 +1495,7 @@ const translations: Readonly> = { }, tdp: { term: '热设计功耗', - aliases: ['TDP', '整卡功耗', '全部包含功耗'], + aliases: ['TDP', '热设计功率包络'], plainEnglish: 'TDP 是芯片设计上可持续消耗并以热量形式散发的功率,是每份加速器规格表上的标题瓦数。', definition: @@ -1504,7 +1505,7 @@ const translations: Readonly> = { significance: '电力已成为 AI 扩建的硬约束,在许多市场甚至排在资本之前。每芯片 TDP 的持续上升迫使行业转向液冷,也让每瓦性能与每美元性能一样,成为比较芯片世代的主要维度。', benchmarkContext: - 'InferenceX 用包含散热和基础设施开销的每芯片全部包含功耗来计算每 token 能耗和每兆瓦 token 数,PowerX 工作流正把这些指标从铭牌数值推向运行中的实测功耗。', + 'Rubin 文章注明所测量产 SKU 的 TDP 为 2300 W,但设施吞吐量采用全口径市电功率归一化。TDP 既不是推理实测功耗,也不是设施总功耗。文中 DSX MaxLPS 部分讨论按工作负载规划供电,并将更细粒度的 PowerX 测量列为后续工作,没有把现有曲线当作实测功耗结果。', }, pue: { term: '电源使用效率', @@ -2308,6 +2309,166 @@ const translations: Readonly> = { benchmarkContext: 'TPU-Sync DRAM offload 和 Mooncake Store 池化被列为 TPU InferenceX 预览的后续步骤,排在 AgentX TPU 结果之前。NVIDIA 和 AMD 的 AgentX 文章已经表明,KV 工作集大小和 offload 容量决定智能体负载的每 token 成本。', }, + 'multi-turn-inference': { + term: '多轮推理', + aliases: ['multi-turn inference', '多轮服务', '多轮工作负载'], + plainEnglish: '多轮会话会连续向模型发送相关请求,把之前的对话和工具结果带入后续提示词。', + definition: '多轮推理为一系列相关的模型请求提供服务,后续输入包含前几轮留下的状态或对话历史。', + explanation: + '智能体会话可能包含数十甚至数百轮。模型输出和工具结果会扩展下一轮提示词,因此大量输入可能已有可复用的缓存状态。工具执行和请求依赖关系还会改变工作到达服务器的时间。', + significance: + '相互独立的提示词序列无法重现这些依赖关系,也无法重现不断增长的缓存工作集。缓存淘汰会导致重复 prefill;即使总吞吐量较高,某次响应过慢仍可能推迟下一轮。', + benchmarkContext: + 'Rubin 文章将多轮结构列为 AgentX 工作负载的主要特征。应在相同智能体场景内比较结果,不能直接套用固定长度、独立请求测试得出的排名。', + }, + 'subagent-bursts': { + term: '子智能体请求突发', + aliases: ['subagent bursts', 'sub-agent bursts', '智能体突发流量'], + plainEnglish: + '智能体同时启动多个短时运行的子任务,会突然增加请求量和服务器需要保存的新上下文。', + definition: + '子智能体请求突发是指主智能体启动多个下级任务,各自发出请求序列,从而在短时间内增加推理需求。', + explanation: + '主会话可能已有很长的可复用前缀,新分支却可能从全新上下文开始。这些分支会产生重叠的 prefill 和 decode 工作,并暂时扩大 KV cache 工作集。分支的启动时间和依赖关系与请求数量同样重要。', + significance: + '较高的平均缓存命中率可能掩盖某些时段大量新增的 prefill 需求。容量规划还需考虑突发时序、缓存淘汰,以及那些必须完成后主任务才能继续的分支延迟。', + benchmarkContext: + 'Rubin 文章将子智能体突发、多轮会话、长上下文和高前缀复用共同列为工作负载特征。因此,AgentX 比较的是随时间变化的请求模式,而不是由相同提示词组成的恒定批次。', + }, + 'p90-interactivity': { + term: 'P90 交互性', + aliases: ['P90 interactivity', 'P90 TPS', 'P90 流式输出速度'], + plainEnglish: 'P90 交互性把较慢尾部的 token 间隔换算成输出速率,数值越高,流式输出越快。', + definition: + '在 Rubin AgentX 分析中,P90 交互性是 P90 全响应 token 间延迟的倒数,单位为每用户每秒 token 数。', + explanation: + '计算时先取延迟的 P90,再取倒数并换算单位。P90 全响应 token 间延迟为 10 毫秒时,对应 100 tok/s/user。这并不是 token 速率的第 90 百分位数,因为正数取倒数后,大小顺序会反转。', + significance: + '分位数和延迟定义共同决定比较采用的速度约束。该指标不包含开始输出前的等待;相同的 P90 交互性仍可能对应明显不同的首 token 时间和端到端延迟。', + benchmarkContext: + 'Rubin 文章在相同 P90 交互性下比较各引擎的性能前沿,只在实测区间内插值。引用吞吐量、成本或功率归一化倍率时,必须保留目标速度与对比引擎。', + measurement: { label: '换算关系', value: 'P90 交互性 = 1000 / P90 全响应 ITL(毫秒)' }, + }, + 'end-to-end-latency': { + term: '端到端延迟', + aliases: ['E2E latency', 'E2EL', '请求完成时间', 'P90 端到端延迟'], + plainEnglish: '端到端延迟是从提交模型请求到收完完整答案的时间,包含开始输出前的等待。', + definition: '请求端到端延迟衡量从提交请求到收到响应最后一个 token 之间的耗时。', + explanation: + '它包含首 token 时间及后续流式输出时长。即使 token 速率相同,更长的答案也需要更多时间,因此比较时须采用可比的输出长度分布。P90 端到端延迟是请求完成时间的第 90 百分位数,不能把各阶段单独计算的 P90 延迟相加得到。', + significance: + '智能体往往需要等到完整响应到达,才能执行工具或开始下一轮。仅有较快的流式输出无法限定这段等待。单次请求延迟也不同于完整智能体任务时长,后者可能包含多次模型调用和工具执行。', + benchmarkContext: + 'Rubin 文章分别绘制 P90 端到端延迟和 P90 交互性。两者结合阅读,才能区分流式输出加快与排队、prefill 或整段响应耗时缩短。', + measurement: { label: '常用单位', value: '每个已完成请求的秒数' }, + }, + 'cached-input-tokens': { + term: '缓存输入 token', + aliases: ['cached input tokens', '缓存提示词 token', '缓存读取 token'], + plainEnglish: + '缓存输入 token 是提示词中可以复用先前处理结果的部分,能减少再次读取同一内容的工作。', + definition: + '缓存输入 token 是通过复用已有模型状态处理的输入 token,无需为其重新执行完整 prefill。', + explanation: + '前几轮对话经常再次出现在后续提示词中。是否命中缓存取决于状态是否保留、请求路由和可用存储。服务商可能对缓存输入、未缓存输入和生成输出分别定价,收入计算也必须区分这三类 token。', + significance: + '缓存 token 仍属于已服务的工作负载,但它不代表与输出 token 相同的新增计算量或销售价值。潜在前缀复用比例、实测缓存命中率和缓存命中的售价,是不同的量。', + benchmarkContext: + 'Rubin 文章在多项比较中使用总 token 吞吐量,并在经济性讨论中区分缓存输入、未缓存输入和输出价格。不能直接将总 token 倍率套用到仅针对输出的价格上估算收入。', + }, + 'billable-utilization': { + term: '可计费利用率', + aliases: ['billable utilization', '可计费容量利用率'], + plainEnglish: '可计费利用率表示在估算期间,有多少建模服务容量实际用于付费流量。', + definition: '可计费利用率是经济模型中假定用于产生收入的流量占可用服务容量的比例。', + explanation: + '基准测试确定某一运行点的 token 速率;换算成年收入时,还需假设这些容量长期能售出多少。容量闲置或需求不足会降低可计费产出,即使服务栈在有负载时能达到实测速率。', + significance: + '这一假设不同于 GPU 利用率遥测或模型 FLOPS 利用率。加速器繁忙并不能证明其工作能够计费,峰值吞吐量测量也不能证明全年都有足够的客户需求。', + benchmarkContext: + 'Rubin 文章的年收入和建模利润示例采用 75 TPS、60% 利用率。引用结果时须同时保留这些假设,不能将其表述为实际收入或纯硬件性能测量。', + measurement: { label: '经济模型假设', value: '估算期间售出的服务容量比例(%)' }, + }, + 'annual-revenue-per-gigawatt': { + term: '每吉瓦年收入', + aliases: ['annual revenue per gigawatt', '每 GW 收入', '每吉瓦市电容量年 token 收入'], + plainEnglish: '这项估算计算在固定一吉瓦数据中心市电容量下,一年的 token 销售可能带来多少收入。', + definition: '每吉瓦年收入是将一年的建模 token 销售收入归一化到一吉瓦全口径市电容量后的指标。', + explanation: + '先在指定交互性目标下,用利用率假设把功率归一化吞吐量换算为全年可计费 token 数;再分别对缓存输入、未缓存输入和输出应用各自价格,最后合计收入。吉瓦表示功率额度,年度则提供时间维度。', + significance: + '该指标把服务效率与电力受限的商业模型联系起来。它不仅取决于实测吞吐量,还取决于需求、价格、token 构成和利用率;扣除成本及适用的许可费后,才能讨论利润。', + benchmarkContext: + 'Rubin 文章在 75 TPS、60% 利用率下展示年收入。每 GW 数值属于归一化估算,不能证明测试实际部署了一吉瓦设备,也不能证明预计 token 数已经售出。', + measurement: { label: '常用单位', value: '美元/市电吉瓦/年' }, + }, + 'modeled-profit-per-gigawatt': { + term: '每吉瓦建模利润', + aliases: ['modeled profit per gigawatt', '每 GW 利润', '每吉瓦市电容量年建模利润'], + plainEnglish: + '该估算从一吉瓦市电容量对应的年度 token 收入中,扣除建模服务成本和适用的模型许可费。', + definition: + '每吉瓦建模利润是在年度 token 收入中扣除模型所包含的成本和许可费,再按一吉瓦全口径市电容量归一化的结果。', + explanation: + '该结果沿用收入模型的交互性、token 价格、缓存构成和利用率假设。计算费用使用选定的自有或租赁成本口径。如果模型许可费按收入比例收取,应基于收入计算,而不是先扣除计算费用后的余额。', + significance: + '这是具有明确口径的经济估算,并非经审计的企业净利润。模型之外的成本会影响实际利润;即使硬件基准性能不变,需求不足或 token 价格下降也会降低收益。', + benchmarkContext: + 'Rubin 文章的 75 TPS 示例采用 60% 利用率,并假设采用 MIT 许可的 DeepSeek V4 Pro 无需模型许可费。将 GW 结果线性缩放到较小部署时,仍沿用这些假设,并非实测的集群利润。', + measurement: { label: '常用单位', value: '建模利润美元/市电吉瓦/年' }, + }, + 'utility-power-budget': { + term: '市电功率预算', + aliases: ['utility power budget', '全口径市电功率', '数据中心电力额度'], + plainEnglish: '市电功率预算是整个数据中心可用的供电容量,包含为服务器供电和制冷的配套设备。', + definition: '市电功率预算是在市电接入边界,为 IT 设备和配套基础设施配置的设施供电容量。', + explanation: + '加速器 TDP 描述组件级设计包络。设施预算还需容纳主机、网络、电力转换、制冷以及口径内的其他开销。在计算能部署多少硬件或提供多少 token 吞吐量之前,必须统一说明系统边界。', + significance: + '如果一个系统只计算芯片功耗,另一个却计算市电功耗,效率比值就会失真。配置容量与运行实测功耗回答的也是不同问题:前者描述部署额度,后者描述特定负载下的消耗。', + benchmarkContext: + 'Rubin 文章按市电 MW 归一化吞吐量,并按全口径市电 GW 归一化年度经济指标。DSX MaxLPS 部分讨论利用工作负载功耗画像,在该额度内增加硬件部署,而不是把 TDP 当作推理实测功耗。', + }, + 'dsx-maxlps': { + term: 'DSX MaxLPS', + aliases: ['NVIDIA DSX MaxLPS', '动态功率调配'], + plainEnglish: + 'DSX MaxLPS 根据工作负载需求管理数据中心功率,以利用保守峰值供电规划下闲置的容量。', + definition: + 'DSX MaxLPS 是 Rubin 文章讨论的 NVIDIA 功率管理方案,用于在受限的数据中心电力额度内动态管理功率。', + explanation: + '如果按所有加速器同时达到峰值功耗规划供电,而推理实际功耗低于设计包络,就会留下闲置容量。文章描述了对当前及代表性未来负载进行功耗分析,再在数据中心内调配功率,从而在现有电力容量内提高部署密度。', + significance: + '可获得的空间取决于实际负载行为和安全的功率控制策略。它不会使供电容量无限增长,也不保证增加加速器后在任何需求模式下都能维持延迟。功耗画像必须覆盖一个有利基准点之外的条件。', + benchmarkContext: + 'Rubin 文章在讨论 MaxLPS 时提到后续 PowerX 集成,但没有在展示的 AgentX 曲线中单独测量 MaxLPS 的加速效果。因此,不能把这些结果解释成该功能贡献的直接测量。', + }, + 'vera-rubin': { + term: 'Vera Rubin', + aliases: ['Vera Rubin 平台', 'Rubin GPU', 'Vera CPU', 'VR NVL72'], + plainEnglish: 'Vera Rubin 是 NVIDIA 的平台,组合 Rubin GPU、Vera CPU 及配套的互连与网络组件。', + definition: + 'Vera Rubin 是 NVIDIA 的平台,文章将其描述为 Rubin GPU、Vera CPU、NVLink 6 Switch、ConnectX-9、BlueField-4 和 Spectrum-6 六款产品的协同设计。', + explanation: + '平台名称涵盖的并不只有加速器。Rubin NVL72 文章评估了采用早期预发布 TensorRT-LLM 软件的完整服务配置,并注明量产 SKU 的 TDP 为 2300 W、每个计算托盘配有 1.5 TB CPU LPDDR5X,不能将所有已公布配置视为相同。', + significance: + '智能体性能取决于计算、内存容量、通信和服务软件的相互作用。实测优势不能只归因于单一组件;机架级结果也不能证明所有模型或延迟目标下都有同样的提升。', + benchmarkContext: + '将 Vera Rubin 与 GB300 或单节点系统比较时,应保留文章的模型、引擎、精度、工作负载及交互性目标。已发布结果对应一个软件快照,文中对后续提升的预期属于预测,而非测量。', + }, + 'extreme-co-design': { + term: '极致协同设计', + aliases: ['extreme co-design', '平台协同设计', '软硬件协同设计'], + plainEnglish: '协同设计把相互依赖的系统组件一起开发,减少其他环节的瓶颈对局部性能提升的抵消。', + definition: + '极致协同设计是 Rubin 文章使用的术语,指围绕目标工作负载,协调开发加速器、主机、互连和网络产品。', + explanation: + '更快的 GPU 仍可能等待内存、通信或请求调度。联合设计这些组件,会改变服务栈能够使用的资源。文章列出了六款协同产品:Rubin GPU、Vera CPU、NVLink 6 Switch、ConnectX-9、BlueField-4 和 Spectrum-6。', + significance: + '这一概念关注完整系统性能,而不只看峰值算力。它是一种架构设计方法,不是基准指标,也不是通过独立实验验证的某个吞吐量倍率的归因。', + benchmarkContext: + 'AgentX 在多轮流量下测量最终软硬件配置。Rubin 文章并未分别改变六款产品来做控制变量实验,因此不能将实测增益分配给各组件,也不能据此确定通用的协同设计加速倍数。', + }, }; const entries = getAllGlossaryEntries().map((entry) => { diff --git a/packages/app/src/lib/glossary.test.ts b/packages/app/src/lib/glossary.test.ts index 88f8db199..ee7a16591 100644 --- a/packages/app/src/lib/glossary.test.ts +++ b/packages/app/src/lib/glossary.test.ts @@ -75,6 +75,57 @@ describe('glossary content', () => { expect(referencedArticles).toEqual(new Set(getAllPosts().map((post) => post.slug))); }); + + it.each([ + 'multi-turn-inference', + 'subagent-bursts', + 'p90-interactivity', + 'end-to-end-latency', + 'cached-input-tokens', + 'billable-utilization', + 'annual-revenue-per-gigawatt', + 'modeled-profit-per-gigawatt', + 'utility-power-budget', + 'dsx-maxlps', + 'vera-rubin', + 'extreme-co-design', + ])( + 'makes Rubin concept %s discoverable in both locales with its article and related terms', + (slug) => { + for (const entry of [getGlossaryEntry(slug), getZhGlossaryEntry(slug)]) { + expect(entry, slug).toBeDefined(); + expect(entry?.articleSlugs).toContain('vera-rubin-nvl72-agentic-inference'); + expect(entry?.relatedTerms.length).toBeGreaterThanOrEqual(3); + } + expect(getZhGlossaryEntry(slug)?.definition).not.toBe(getGlossaryEntry(slug)?.definition); + }, + ); + + it('separates percentile streaming speed from response completion and normalized interactivity', () => { + const p90 = getGlossaryEntry('p90-interactivity')!; + expect(p90.measurement?.value).toContain('1000 / P90 full-response ITL (ms)'); + expect(p90.relatedTerms).toContain('end-to-end-latency'); + expect(getGlossaryEntry('interactivity')?.relatedTerms).toContain(p90.slug); + expect(getGlossaryEntry('end-to-end-latency')?.relatedTerms).toContain( + 'e2e-normalized-interactivity', + ); + expect(getZhGlossaryEntry(p90.slug)?.measurement?.value).toContain( + '1000 / P90 全响应 ITL(毫秒)', + ); + }); + + it('keeps NVL72 platform generations and component versus facility power distinct', () => { + for (const lookup of [getGlossaryEntry, getZhGlossaryEntry]) { + const nvl72 = lookup('nvl72')!; + expect(nvl72.aliases).toContain('Vera Rubin NVL72'); + expect(nvl72.explanation).toContain('NVLink 5'); + expect(nvl72.explanation).toContain('NVLink 6'); + expect(lookup('vera-rubin')?.relatedTerms).toContain(nvl72.slug); + expect(lookup('utility-power-budget')?.relatedTerms).toContain('tdp'); + } + expect(getGlossaryEntry('tdp')?.aliases).not.toContain('all-in power'); + expect(getZhGlossaryEntry('tdp')?.aliases).not.toContain('全部包含功耗'); + }); }); describe('Chinese glossary content', () => { diff --git a/packages/app/src/lib/glossary.ts b/packages/app/src/lib/glossary.ts index 7f8e2135f..1bddc5b6e 100644 --- a/packages/app/src/lib/glossary.ts +++ b/packages/app/src/lib/glossary.ts @@ -236,16 +236,16 @@ const entries = [ plainEnglish: 'Throughput is how much total work the system gets done each second across everyone using it.', definition: - 'Throughput is the total rate at which an inference system produces tokens across all active requests.', + 'Throughput is the rate of token processing across requests. Output throughput counts generated tokens; total token throughput counts input and output tokens under the stated benchmark accounting.', explanation: 'InferenceX commonly normalizes throughput as tokens per second per chip so systems of different sizes can be compared. Higher batching or concurrency often raises aggregate throughput because weight reads and compute are amortized across more requests, but individual users may receive tokens more slowly.', significance: 'Maximum throughput captures only one operating point. A system can lead in tokens per second while operating at interactivity too low for a real-time product. The useful comparison is throughput at a latency or interactivity target appropriate to the workload.', benchmarkContext: - 'On an InferenceX chart, throughput is read together with interactivity across the full concurrency sweep. The Pareto frontier removes operating points that are worse on both axes.', + 'On an InferenceX chart, throughput is read together with interactivity across the full concurrency sweep. The Rubin AgentX article reports total token throughput, including reused input, rather than only newly generated output. Check the token accounting before comparing that curve with an output-only result.', measurement: { label: 'Typical unit', value: 'tokens/second/chip (tok/s/chip)' }, relatedTerms: ['interactivity', 'concurrency', 'pareto-frontier', 'iso-interactivity'], - articleSlugs: [INFERENCEMAX, INFERENCEX_V2, SGLANG_056], + articleSlugs: [INFERENCEMAX, INFERENCEX_V2, SGLANG_056, VR_RUBIN_AGENTIC], }, { slug: 'interactivity', @@ -261,7 +261,7 @@ const entries = [ significance: 'Different products need different operating points. Voice and interactive coding demand high token rates, while offline summarization can trade interactivity for much more aggregate throughput. Comparing hardware at unmatched interactivity can therefore produce a misleading winner.', benchmarkContext: - 'InferenceX plots tokens per second per user against throughput or cost. Iso-interactivity tables interpolate each system’s Pareto frontier at the same token rate so the comparison holds user experience constant. Because this axis ignores the wait before the first token, agentic charts also offer E2E Normalized Interactivity, which folds TTFT into the same unit.', + 'InferenceX plots tokens per second per user against throughput or cost. The Rubin AgentX article uses the reciprocal of P90 full-response inter-token latency for P90 interactivity. Matching that speed does not match TTFT or end-to-end latency. E2E Normalized Interactivity is a separate metric that includes the initial wait.', measurement: { label: 'Typical unit', value: 'tokens/second/user (tok/s/user)' }, relatedTerms: [ 'time-per-output-token', @@ -269,8 +269,10 @@ const entries = [ 'iso-interactivity', 'e2e-normalized-interactivity', 'latency', + 'p90-interactivity', + 'end-to-end-latency', ], - articleSlugs: [INFERENCEMAX, INFERENCEX_V2, MI355X_KIMI, TILERT], + articleSlugs: [INFERENCEMAX, INFERENCEX_V2, MI355X_KIMI, TILERT, VR_RUBIN_AGENTIC], }, { slug: 'latency', @@ -399,11 +401,18 @@ const entries = [ explanation: 'Benchmark runs rarely land at identical tok/s/user values because each recipe has different concurrency points. An iso-interactivity comparison interpolates each Pareto frontier at a shared target and then compares throughput, cost, or efficiency there.', significance: - 'Holding user experience constant avoids a common benchmark error: declaring a high-throughput system faster when it reaches that throughput only by serving every request more slowly.', + 'Matching streaming speed avoids declaring a system faster merely because it serves more requests at a slower token rate. It does not hold the entire user experience constant: initial waiting time and total response duration still need separate comparisons.', benchmarkContext: 'InferenceX articles use iso-interactivity tables for hardware, precision, and software comparisons. Values outside a measured frontier are marked unreachable and are not extrapolated beyond observed data. The frontier is always built on throughput against interactivity; cost per million tokens and joules per token are then derived from the interpolated throughput rather than splined on their own, because each is a per-chip constant divided by that throughput and splining it separately would break the identity between knots.', relatedTerms: ['interactivity', 'pareto-frontier', 'throughput', 'performance-per-dollar'], - articleSlugs: [B200_GLM5, B200_MINIMAX, B200_KIMI, GB300_DSV4, AGENTX_GLM_SGLANG], + articleSlugs: [ + B200_GLM5, + B200_MINIMAX, + B200_KIMI, + GB300_DSV4, + AGENTX_GLM_SGLANG, + VR_RUBIN_AGENTIC, + ], }, { slug: 'input-output-sequence-length', @@ -465,7 +474,7 @@ const entries = [ significance: 'Peak chip FLOPS account for only part of serving economics. Memory, networking, software maturity, numerical precision, and achievable utilization all affect the measured output behind the ratio.', benchmarkContext: - 'InferenceX compares infrastructure perf/$ at matched interactivity and names the TCO inputs used. Ratios should not be carried across different model, sequence-length, precision, or latency regimes. Cost per million tokens and the total, input, and output infrastructure purchasing-power axes express those TCO economics.', + 'InferenceX compares infrastructure perf/$ at matched interactivity and names the TCO inputs used. The Rubin article’s approximately 67x result is scoped to 170 TPS, owning costs, and the specified TRTLLM NVFP4 Dense configurations. It is not a generation-wide multiplier: the ratio changes with the target, comparison engine, and owning versus rental cost basis.', relatedTerms: [ 'cost-per-million-tokens', 'tokens-per-dollar', @@ -480,6 +489,7 @@ const entries = [ MI355X_GLM5, AGENTX_DSV4_MI355X_B200, TPU_IRONWOOD, + VR_RUBIN_AGENTIC, ], }, { @@ -496,14 +506,14 @@ const entries = [ significance: 'Using TCO instead of list price makes cross-system economics more realistic, especially for rack-scale products whose networking and power infrastructure differ. The result remains a model and should be read with its assumptions.', benchmarkContext: - 'InferenceX combines SemiAnalysis AI Cloud TCO inputs with observed tok/s/chip. This separates hourly system cost from the software and workload behavior that determines how many tokens that hour produces.', + 'InferenceX combines SemiAnalysis AI Cloud TCO inputs with observed tok/s/chip. The Rubin article distinguishes owning at large hyperscaler volume, including amortized hardware, networking, facilities, power, and capital costs, from the customer price of a three-year cloud reservation. These are alternative cost bases, not costs to add together; rental pricing also reflects the provider’s commercial terms.', relatedTerms: [ 'cost-per-million-tokens', 'performance-per-dollar', 'tokens-per-megawatt', 'throughput', ], - articleSlugs: [INFERENCEMAX, INFERENCEX_V2, GB200_R1, VR_RUBIN, JALAPENO], + articleSlugs: [INFERENCEMAX, INFERENCEX_V2, GB200_R1, VR_RUBIN, JALAPENO, VR_RUBIN_AGENTIC], }, { slug: 'tokens-per-megawatt', @@ -519,7 +529,7 @@ const entries = [ significance: 'Power availability is often the binding constraint on new AI deployments. A system that produces more tokens per provisioned megawatt can serve more demand from the same utility allocation even if its individual accelerators draw more power.', benchmarkContext: - 'Compare tokens/MW at the same model, workload shape, precision, and interactivity. Otherwise a high-throughput low-interactivity point can appear efficient while failing the target user experience. Energy per token expresses the same provisioned budget per unit of output, and InferenceX additionally reports measured accelerator energy where the telemetry is trustworthy.', + 'The Rubin article compares total tok/s per utility MW at matched P90 interactivity, with interpolation only inside each engine’s measured range. Preserve the engine, precision, caching, and parallelism labels when quoting a ratio. The provisioned utility-power denominator is a capacity model, distinct from measured accelerator power or energy telemetry.', measurement: { label: 'Typical unit', value: 'tokens/second per provisioned utility MW' }, relatedTerms: [ 'throughput', @@ -527,8 +537,10 @@ const entries = [ 'interactivity', 'total-cost-of-ownership', 'performance-per-dollar', + 'utility-power-budget', + 'annual-revenue-per-gigawatt', ], - articleSlugs: [INFERENCEMAX, DEEPSEEK_V4, VR_RUBIN, JALAPENO], + articleSlugs: [INFERENCEMAX, DEEPSEEK_V4, VR_RUBIN, JALAPENO, VR_RUBIN_AGENTIC], }, { slug: 'prefill', @@ -607,7 +619,7 @@ const entries = [ definition: 'Prefix caching reuses KV-cache state when multiple requests begin with the same token sequence.', explanation: - 'A repeated system prompt, shared document, or common conversation prefix can reuse cached states. A cache hit can reduce prompt computation and time to first token.', + 'A repeated system prompt, shared document, or common conversation prefix can reuse cached states. In a growing agent session, earlier outputs become part of later inputs, increasing potential prefix reuse. Actual hits still depend on the state remaining available and reachable; eviction or fresh subagent contexts can require new prefill.', significance: 'Production workloads with repeated prefixes may outperform synthetic random-token benchmarks. The benefit depends on hit rate, cache capacity, eviction policy, and whether requests route to workers that hold the needed state.', benchmarkContext: @@ -619,8 +631,17 @@ const entries = [ 'prefill', 'time-to-first-token', 'nvidia-dynamo', + 'multi-turn-inference', + 'cached-input-tokens', + ], + articleSlugs: [ + AGENTIC_WORKLOADS, + INFERENCEX_V2, + GB200_KIMI, + KIMI_K3, + AGENTX_V3, + VR_RUBIN_AGENTIC, ], - articleSlugs: [AGENTIC_WORKLOADS, INFERENCEX_V2, GB200_KIMI, KIMI_K3, AGENTX_V3], }, { slug: 'disaggregated-inference', @@ -890,13 +911,13 @@ const entries = [ definition: 'NVLink is NVIDIA’s high-bandwidth accelerator interconnect for moving data directly among chips within a scale-up domain.', explanation: - 'NVSwitch systems connect multiple NVLink endpoints so collectives can span an eight-chip server or, in NVL72 products, a 72-chip rack-scale domain. That bandwidth is distinct from the InfiniBand or Ethernet fabric connecting separate systems.', + 'NVSwitch systems connect multiple NVLink endpoints so collectives can span a node or a 72-chip rack-scale domain. The generation depends on the platform: Blackwell NVL72 uses NVLink 5, while the Rubin article identifies NVLink 6 Switch as part of Vera Rubin. This scale-up fabric is distinct from the networking used between systems.', significance: 'Large TP and especially wide-EP groups exchange data at every generated token. Keeping those collectives on NVLink can make a rack-scale recipe faster than a similar chip count spread across scale-out links.', benchmarkContext: 'InferenceX compares both node-level chips and NVL72 systems. Interpret the system topology and parallel group width before attributing the entire result to per-chip compute.', relatedTerms: ['scale-up-vs-scale-out', 'all-to-all', 'all-reduce', 'wide-expert-parallelism'], - articleSlugs: [GB200_R1, GB200_KIMI, INFERENCEX_V2, VR_RUBIN], + articleSlugs: [GB200_R1, GB200_KIMI, INFERENCEX_V2, VR_RUBIN, VR_RUBIN_AGENTIC], }, { slug: 'quantization', @@ -1773,14 +1794,14 @@ const entries = [ { slug: 'nvl72', term: 'NVL72', - aliases: ['GB200 NVL72', 'GB300 NVL72', 'rack-scale system'], + aliases: ['GB200 NVL72', 'GB300 NVL72', 'Vera Rubin NVL72', 'rack-scale system'], category: 'Hardware', plainEnglish: 'NVL72 is a rack where 72 accelerators share one high-speed fabric, so they behave more like a single large machine than a cluster.', definition: 'NVL72 is a rack-scale NVIDIA system that places 72 accelerators in a single NVLink scale-up domain rather than in separate eight-chip nodes.', explanation: - 'The dashboard specs record NVLink 5.0 at 900 GB/s per chip unidirectional across a scale-up world size of 72, switched through NVSwitch. A conventional node keeps that bandwidth among eight chips and falls back to slower scale-out networking beyond them, so the difference is not raw speed but how many chips are reachable before the fabric changes character.', + 'NVL72 describes the size of the NVLink domain, not one fixed chip generation or bandwidth. GB200 and GB300 systems use Blackwell-generation hardware and NVLink 5; Vera Rubin pairs Rubin GPUs with Vera CPUs and NVLink 6 Switch. Compare the specific platform rather than applying Blackwell specifications to every NVL72 rack.', significance: 'Techniques whose cost is dominated by collectives change economics inside a large domain. Wide expert parallelism spreads experts across many chips and pays all-to-all traffic for every token, which is tolerable at scale-up bandwidth and often is not across a scale-out fabric.', benchmarkContext: @@ -1792,7 +1813,15 @@ const entries = [ 'all-to-all', 'total-cost-of-ownership', ], - articleSlugs: [GB200_R1, GB300_DSV4, GB200_KIMI, VR_RUBIN, AGENTX_K3_ATOM, JALAPENO], + articleSlugs: [ + GB200_R1, + GB300_DSV4, + GB200_KIMI, + VR_RUBIN, + AGENTX_K3_ATOM, + JALAPENO, + VR_RUBIN_AGENTIC, + ], }, { slug: 'atom', @@ -2330,7 +2359,7 @@ const entries = [ slug: 'tdp', term: 'Thermal design power', abbreviation: 'TDP', - aliases: ['TDP', 'board power', 'all-in power'], + aliases: ['TDP', 'thermal power envelope'], category: 'Hardware', plainEnglish: 'TDP is the sustained power a chip is designed to draw and shed as heat, the headline wattage on every accelerator spec sheet.', @@ -2341,9 +2370,9 @@ const entries = [ significance: 'Power has become the binding constraint of AI buildout, ahead of capital in many markets. Rising per chip TDP forced the shift to liquid cooling and made performance per watt, not just performance per dollar, a primary axis for comparing silicon generations.', benchmarkContext: - 'InferenceX derives energy per token and tokens per megawatt using per chip all in power figures that include cooling and infrastructure overhead above TDP, and the PowerX workstream is extending this from rated figures toward measured draw during runs.', + 'The Rubin article specifies a 2300 W TDP production SKU but normalizes facility throughput with all-in utility power. TDP is neither measured inference draw nor total facility power. Its DSX MaxLPS discussion concerns workload-aware power provisioning; the article describes finer-grained PowerX measurements as upcoming rather than treating the displayed curves as measured-power results.', relatedTerms: ['energy-per-token', 'tokens-per-megawatt', 'pue', 'total-cost-of-ownership'], - articleSlugs: [INFERENCEX_V2, VR_RUBIN], + articleSlugs: [INFERENCEX_V2, VR_RUBIN, VR_RUBIN_AGENTIC], }, { slug: 'pue', @@ -3548,6 +3577,265 @@ const entries = [ ], articleSlugs: [TPU_IRONWOOD, AGENTX_V3, AGENTIC_WORKLOADS], }, + { + slug: 'multi-turn-inference', + term: 'Multi-turn inference', + aliases: ['multi-turn serving', 'multi-turn workload'], + category: 'Agentic inference', + plainEnglish: + 'A multi-turn session sends several related requests to the model, carrying earlier conversation and tool results into later prompts.', + definition: + 'Multi-turn inference serves a sequence of related model requests whose inputs include state or conversation history from earlier turns.', + explanation: + 'In agentic workloads, a session may span tens or hundreds of turns. Generated output and tool results extend the next prompt, so much of its input may already have cached state. Tool execution and dependent requests also change when work reaches the server.', + significance: + 'A sequence of independent prompts does not reproduce these dependencies or the growing cache working set. Cache eviction can force repeated prefill, while a slow response can delay the next turn even when aggregate throughput appears high.', + benchmarkContext: + 'The Rubin article identifies multi-turn structure as a defining AgentX workload property. Compare these results with the same agentic scenario rather than transferring rankings from fixed-length, independent-request tests.', + relatedTerms: ['agentx', 'trace-replay', 'prefix-caching', 'long-context', 'subagent-bursts'], + articleSlugs: [VR_RUBIN_AGENTIC, AGENTX_V3], + }, + { + slug: 'subagent-bursts', + term: 'Subagent bursts', + aliases: ['sub-agent bursts', 'bursty agent traffic'], + category: 'Agentic inference', + plainEnglish: + 'An agent can launch several short-lived helpers together, suddenly adding requests and new context for the server to hold.', + definition: + 'Subagent bursts are short periods of increased inference demand caused by an agent launching multiple subordinate tasks with their own request sequences.', + explanation: + 'A parent session can have a long reusable prefix while its new branches start with fresh context. These branches create overlapping prefill and decode work and temporarily enlarge the KV-cache working set. Their start times and dependencies matter as much as their request count.', + significance: + 'A high average cache-hit rate can hide periods of heavy new-prefill demand. Capacity planning must account for burst timing, cache eviction, and the latency of branches whose completion blocks the parent task.', + benchmarkContext: + 'The Rubin article names subagent bursts alongside multi-turn sessions, long context, and high prefix reuse. AgentX comparisons therefore concern a time-varying request pattern, not a constant batch of identical prompts.', + relatedTerms: ['subagent', 'multi-turn-inference', 'concurrency', 'kv-cache', 'prefill'], + articleSlugs: [VR_RUBIN_AGENTIC, AGENTIC_WORKLOADS], + }, + { + slug: 'p90-interactivity', + term: 'P90 interactivity', + aliases: ['P90 TPS', 'P90 token rate', 'P90 streaming speed'], + category: 'Benchmark metrics', + plainEnglish: + 'P90 interactivity expresses a slower-tail streaming interval as a token rate, so higher values mean faster response streaming.', + definition: + 'In the Rubin AgentX analysis, P90 interactivity is the reciprocal of P90 full-response inter-token latency, expressed in tokens per second per user.', + explanation: + 'Take the P90 latency statistic first, then invert it with the appropriate unit conversion. A P90 full-response inter-token latency of 10 milliseconds corresponds to 100 tok/s/user. This is not the 90th percentile of token rates: reciprocation reverses the ordering of positive values.', + significance: + 'The percentile and latency definition determine what speed target a comparison enforces. This metric excludes the initial wait before streaming; equal P90 interactivity can coexist with very different time to first token and end-to-end latency.', + benchmarkContext: + 'The Rubin article compares engine-specific frontiers at matched P90 interactivity and interpolates only within measured ranges. Preserve the target and the comparison engine when citing its throughput, cost, or power ratios.', + measurement: { + label: 'Relationship', + value: 'P90 interactivity = 1000 / P90 full-response ITL (ms)', + }, + relatedTerms: [ + 'interactivity', + 'time-per-output-token', + 'iso-interactivity', + 'end-to-end-latency', + ], + articleSlugs: [VR_RUBIN_AGENTIC], + }, + { + slug: 'end-to-end-latency', + term: 'End-to-end latency', + abbreviation: 'E2E latency', + aliases: ['E2EL', 'request completion time', 'P90 E2E latency'], + category: 'Benchmark metrics', + plainEnglish: + 'End-to-end latency is the time from sending a model request until its complete answer has arrived, including the initial wait.', + definition: + 'Request end-to-end latency measures elapsed time from request submission to receipt of the final response token.', + explanation: + 'It includes time to first token and the subsequent streaming duration. Longer answers take longer even at the same token rate, so comparisons need compatible output-length distributions. P90 E2E latency is the 90th percentile of request completion times, not a sum of separately calculated P90 stage latencies.', + significance: + 'Agents often wait for a complete response before executing a tool or starting a dependent turn. Fast streaming alone does not bound that wait. Request latency also differs from the duration of a whole agent task, which can include many model calls and tool executions.', + benchmarkContext: + 'The Rubin article plots P90 E2E latency separately from P90 interactivity. Read both to distinguish improved streaming cadence from reduced queueing, prefill, or overall response time.', + measurement: { label: 'Typical unit', value: 'seconds per completed request' }, + relatedTerms: [ + 'latency', + 'time-to-first-token', + 'p90-interactivity', + 'e2e-normalized-interactivity', + ], + articleSlugs: [VR_RUBIN_AGENTIC], + }, + { + slug: 'cached-input-tokens', + term: 'Cached input tokens', + aliases: ['cached prompt tokens', 'cache-read tokens'], + category: 'Serving', + plainEnglish: + 'Cached input tokens are parts of a prompt whose earlier processing can be reused, reducing the work needed to read the prompt again.', + definition: + 'Cached input tokens are input tokens served using reusable cached model state rather than recomputing their full prefill.', + explanation: + 'Earlier conversation turns often reappear in later prompts. Whether those tokens hit the cache depends on retained state, routing, and available storage. A provider may price cached input separately from uncached input and generated output, so the three token classes must remain separate in revenue calculations.', + significance: + 'A cached token still belongs to the served workload but does not imply the same new computation or sales value as an output token. Potential prefix reuse, measured cache-hit rate, and the price charged for a cache hit describe different quantities.', + benchmarkContext: + 'The Rubin article uses total token throughput for several comparisons and distinguishes cached-input, uncached-input, and output prices in its economics discussion. A total-token multiplier cannot be applied directly to an output-only price to estimate revenue.', + relatedTerms: [ + 'prefix-caching', + 'prefix-cache-hit-rate', + 'throughput', + 'annual-revenue-per-gigawatt', + ], + articleSlugs: [VR_RUBIN_AGENTIC, AGENTX_V3], + }, + { + slug: 'billable-utilization', + term: 'Billable utilization', + aliases: ['billable capacity utilization', 'revenue utilization'], + category: 'Benchmark metrics', + plainEnglish: + 'Billable utilization is the share of modeled serving capacity that is actually used for paid traffic over the period being estimated.', + definition: + 'Billable utilization is the fraction of available serving capacity assumed to produce revenue-generating traffic in an economic model.', + explanation: + 'A benchmark establishes a token rate at an operating point. Annualization then needs an assumption about how much of that capacity can be sold over time. Idle capacity and insufficient demand reduce billable output even if the serving stack can achieve its measured rate when loaded.', + significance: + 'This assumption is distinct from GPU utilization telemetry or model FLOPS utilization. A busy accelerator does not prove that its work is billable, and a measured peak-throughput result does not establish year-round customer demand.', + benchmarkContext: + 'The Rubin article’s annual revenue and modeled profit example uses 60% utilization at 75 TPS. Keep that assumption with the result rather than presenting the estimate as realized revenue or a hardware-only performance measurement.', + measurement: { + label: 'Economic assumption', + value: 'share of serving capacity sold over time (%)', + }, + relatedTerms: [ + 'gpu-utilization', + 'model-flops-utilization', + 'annual-revenue-per-gigawatt', + 'modeled-profit-per-gigawatt', + ], + articleSlugs: [VR_RUBIN_AGENTIC], + }, + { + slug: 'annual-revenue-per-gigawatt', + term: 'Annual revenue per gigawatt', + aliases: ['revenue per GW', 'annual token revenue per utility GW'], + category: 'Benchmark metrics', + plainEnglish: + 'This estimate asks how much token sales could earn in a year from a fixed gigawatt of datacenter utility power.', + definition: + 'Annual revenue per gigawatt is modeled token-sales revenue over one year, normalized to one gigawatt of all-in utility power capacity.', + explanation: + 'At a chosen interactivity target, convert power-normalized throughput into annual billable token volumes using the utilization assumption. Apply the respective prices to cached input, uncached input, and output volumes, then sum their revenue. A gigawatt is a power allocation; the year supplies the time dimension.', + significance: + 'The result connects serving efficiency to a power-constrained business model. It depends on demand, prices, token mix, and utilization as well as measured throughput, and it does not show profit until costs and any license fees are deducted.', + benchmarkContext: + 'The Rubin article presents annual revenue at 75 TPS and 60% utilization. Its per-GW values are normalized estimates, not evidence that a full gigawatt deployment was benchmarked or that the projected token volumes were sold.', + measurement: { label: 'Typical unit', value: 'USD per utility GW per year' }, + relatedTerms: [ + 'tokens-per-megawatt', + 'billable-utilization', + 'cached-input-tokens', + 'modeled-profit-per-gigawatt', + 'utility-power-budget', + ], + articleSlugs: [VR_RUBIN_AGENTIC], + }, + { + slug: 'modeled-profit-per-gigawatt', + term: 'Modeled profit per gigawatt', + aliases: ['profit per GW', 'annual modeled profit per utility GW'], + category: 'Benchmark metrics', + plainEnglish: + 'This estimate subtracts modeled serving costs and applicable model-license fees from annual token revenue for a gigawatt of utility power.', + definition: + 'Modeled profit per gigawatt is annual token revenue less the costs and license fees included in the model, normalized to an all-in utility gigawatt.', + explanation: + 'The result inherits the revenue model’s interactivity, token prices, cache mix, and utilization assumptions. Compute expense uses the selected owning or rental cost basis. If a license fee is specified as a percentage of revenue, it is calculated on revenue rather than on the amount left after compute costs.', + significance: + 'This is a defined economic estimate, not audited corporate net income. Costs outside the model can change realized profit, and weak demand or falling token prices can reduce earnings without changing the benchmark performance of the hardware.', + benchmarkContext: + 'The Rubin article’s 75 TPS example assumes 60% utilization and no model-license fee for MIT-licensed DeepSeek V4 Pro. Its linear conversion from a GW to a smaller deployment holds the same assumptions; it is not a measured fleet-scale profit result.', + measurement: { label: 'Typical unit', value: 'modeled USD profit per utility GW per year' }, + relatedTerms: [ + 'annual-revenue-per-gigawatt', + 'billable-utilization', + 'total-cost-of-ownership', + 'utility-power-budget', + ], + articleSlugs: [VR_RUBIN_AGENTIC], + }, + { + slug: 'utility-power-budget', + term: 'Utility power budget', + aliases: ['all-in utility power', 'provisioned utility power', 'datacenter power allocation'], + category: 'Hardware', + plainEnglish: + 'A utility power budget is the electricity capacity available to the whole datacenter, including the equipment that powers and cools the servers.', + definition: + 'A utility power budget is the provisioned facility power capacity available for IT equipment and supporting infrastructure at the utility boundary.', + explanation: + 'Accelerator TDP covers a component-level design envelope. A facility budget must also accommodate hosts, networking, power conversion, cooling, and other included overhead. The system boundary must be stated consistently before using the budget to calculate how much hardware or token throughput fits.', + significance: + 'Comparing one system at chip-only power with another at utility power biases the efficiency ratio. Provisioned capacity and measured operating draw also answer different questions: one describes the deployment allocation, while the other describes consumption during a specific workload.', + benchmarkContext: + 'The Rubin article normalizes throughput per utility MW and annual economics per all-in utility GW. Its DSX MaxLPS discussion considers using workload power profiles to fit more hardware within that allocation without equating TDP to measured inference power.', + relatedTerms: ['tdp', 'pue', 'tokens-per-megawatt', 'dsx-maxlps'], + articleSlugs: [VR_RUBIN_AGENTIC], + }, + { + slug: 'dsx-maxlps', + term: 'DSX MaxLPS', + aliases: ['NVIDIA DSX MaxLPS', 'dynamic power shifting'], + category: 'Software', + plainEnglish: + 'DSX MaxLPS manages datacenter power around workload demand so operators can use capacity that conservative peak-power provisioning would leave unused.', + definition: + 'DSX MaxLPS is NVIDIA’s power-management approach discussed in the Rubin article for dynamically managing power within a constrained datacenter allocation.', + explanation: + 'Provisioning every accelerator for simultaneous peak draw can leave unused capacity when inference workloads consume less than their design envelopes. The article describes profiling current and representative future workloads, then steering power across the datacenter to support a denser deployment within the available power footprint.', + significance: + 'The opportunity depends on actual workload behavior and safe power-control policies. It does not make utility capacity unlimited or guarantee that adding accelerators will preserve latency under every demand pattern. Workload profiles must cover conditions beyond one favorable benchmark point.', + benchmarkContext: + 'The Rubin article describes MaxLPS alongside an upcoming PowerX integration. It does not isolate a measured MaxLPS speedup in the displayed AgentX curves, so those results should not be presented as a direct measurement of this feature’s contribution.', + relatedTerms: ['utility-power-budget', 'tdp', 'tokens-per-megawatt', 'energy-per-token'], + articleSlugs: [VR_RUBIN_AGENTIC], + }, + { + slug: 'vera-rubin', + term: 'Vera Rubin', + aliases: ['Vera Rubin platform', 'Rubin GPU', 'Vera CPU', 'VR NVL72'], + category: 'Hardware', + plainEnglish: + 'Vera Rubin is NVIDIA’s platform combining Rubin GPUs, Vera CPUs, and the interconnect and networking components around them.', + definition: + 'Vera Rubin is the NVIDIA platform that the article describes as co-designed across Rubin GPU, Vera CPU, NVLink 6 Switch, ConnectX-9, BlueField-4, and Spectrum-6.', + explanation: + 'The platform name covers more than an accelerator alone. The Rubin NVL72 article evaluates a complete serving configuration with early pre-release TensorRT-LLM software. It specifies a production SKU with 2300 W TDP and 1.5 TB of CPU LPDDR5X per compute tray, rather than treating every announced configuration as identical.', + significance: + 'Agentic performance reflects the interaction of compute, memory capacity, communication, and serving software. A measured advantage cannot be assigned solely to one component, and a rack-level result does not establish an identical advantage for every model or latency target.', + benchmarkContext: + 'Use the article’s model, engine, precision, workload, and interactivity target when comparing Vera Rubin with GB300 or single-node systems. The published results are a software snapshot; the article’s expectations for later gains are projections rather than measurements.', + relatedTerms: ['nvl72', 'nvlink', 'extreme-co-design', 'agentx', 'tensorrt-llm'], + articleSlugs: [VR_RUBIN_AGENTIC, VR_RUBIN], + }, + { + slug: 'extreme-co-design', + term: 'Extreme co-design', + aliases: ['platform co-design', 'hardware-software co-design'], + category: 'Hardware', + plainEnglish: + 'Co-design means developing connected parts of the system together so a gain in one part is not lost to a bottleneck elsewhere.', + definition: + 'Extreme co-design is the Rubin article’s term for coordinated platform development across accelerator, host, interconnect, and networking products to serve the target workload.', + explanation: + 'A faster GPU can still wait for memory, communication, or request scheduling. Designing these components together changes the resources available to the serving stack. The article names six coordinated products: Rubin GPU, Vera CPU, NVLink 6 Switch, ConnectX-9, BlueField-4, and Spectrum-6.', + significance: + 'The concept directs attention to whole-system performance rather than peak compute alone. It is an architectural approach, not a benchmark metric or an experimentally isolated explanation for a particular throughput multiplier.', + benchmarkContext: + 'AgentX measures the resulting hardware-and-software configuration under multi-turn traffic. The Rubin article’s comparison does not independently vary all six products, so it cannot allocate the measured gain to each component or establish a universal co-design speedup.', + relatedTerms: ['vera-rubin', 'nvlink', 'memory-bandwidth', 'inference-engine', 'agentx'], + articleSlugs: [VR_RUBIN_AGENTIC], + }, ] as const satisfies readonly GlossaryEntry[]; export type GlossaryPreview = Pick< diff --git a/packages/app/src/lib/inference-labels.test.ts b/packages/app/src/lib/inference-labels.test.ts index 0b2b64c6c..25658e580 100644 --- a/packages/app/src/lib/inference-labels.test.ts +++ b/packages/app/src/lib/inference-labels.test.ts @@ -1,17 +1,74 @@ -import { describe, expect, it } from 'vitest'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; import { getHardwareConfig } from '@/lib/constants'; import { getInferenceHardwareConfig, getInferenceRunLabel, + getOverlayLineLabel, getPointHardwareConfig, inferenceFrameworkLabelOverride, + shortRunTag, } from './inference-labels'; const runUrl = 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34926284365'; const historicalUrl = 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34926284364'; const hwKey = 'mi355x_mori-sglang'; +describe('temporary DSpark UMBP run label', () => { + const dsparkUrl = 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/35166686551'; + const expiresAt = Date.parse('2026-10-09T01:32:00Z'); + + beforeEach(() => { + vi.spyOn(Date, 'now').mockReturnValue(Date.parse('2026-09-18T01:32:00Z')); + }); + afterEach(() => vi.restoreAllMocks()); + + it.each([dsparkUrl, `${dsparkUrl}/attempts/1`, `${dsparkUrl}/attempts/5`])( + 'labels the new run without changing identity: %s', + (url) => { + expect(inferenceFrameworkLabelOverride('mori-sglang', url)).toBe('MoRI UMBP SGLang'); + expect(inferenceFrameworkLabelOverride('sglang-disagg', url)).toBe('MoRI UMBP SGLang'); + expect(inferenceFrameworkLabelOverride('sglang', url)).toBeUndefined(); + const config = getInferenceHardwareConfig(hwKey, undefined, [{ run_url: url }]); + expect(config.suffix).toBe('(MoRI UMBP SGLang)'); + expect(config.name).toBe(getHardwareConfig(hwKey).name); + expect(getPointHardwareConfig({ hwKey, run_url: url }, config)).toEqual(config); + expect(getInferenceRunLabel('✕ dspark', [{ framework: 'mori-sglang', run_url: url }])).toBe( + '✕ dspark (MoRI UMBP SGLang)', + ); + }, + ); + + it('leaves neighboring run IDs and mixed historical points correctly labeled', () => { + for (const id of ['35166686550', '35166686552', '351666865510']) { + expect( + inferenceFrameworkLabelOverride('mori-sglang', dsparkUrl.replace('35166686551', id)), + ).toBeUndefined(); + } + expect( + getInferenceHardwareConfig(hwKey, undefined, [ + { run_url: dsparkUrl }, + { run_url: historicalUrl }, + ]).suffix, + ).toBe('(MoRI SGLang / MoRI UMBP SGLang)'); + }); + + it('returns to the standard label at the exact cutoff in all shared display paths', () => { + vi.mocked(Date.now).mockReturnValue(expiresAt - 1); + expect(inferenceFrameworkLabelOverride('mori-sglang', dsparkUrl)).toBe('MoRI UMBP SGLang'); + for (const now of [expiresAt, expiresAt + 1, expiresAt + 86_400_000]) { + vi.mocked(Date.now).mockReturnValue(now); + expect(inferenceFrameworkLabelOverride('mori-sglang', dsparkUrl)).toBeUndefined(); + const point = { hwKey, framework: 'mori-sglang', run_url: dsparkUrl }; + const generic = getHardwareConfig(hwKey); + expect(getInferenceHardwareConfig(hwKey, undefined, [point])).toBe(generic); + expect(getPointHardwareConfig(point, generic)).toBe(generic); + expect(getInferenceRunLabel('✕ dspark', [point])).toBe('✕ dspark'); + expect(inferenceFrameworkLabelOverride('mori-sglang', runUrl)).toBe('MoRI UMBP SGLang'); + } + }); +}); + describe('run-specific MoRI UMBP labels', () => { it('retains the unofficial marker and branch while labeling only the target overlay', () => { expect( @@ -68,3 +125,36 @@ describe('run-specific MoRI UMBP labels', () => { expect(getInferenceHardwareConfig(hwKey, undefined, []).suffix).toBe('(MoRI SGLang)'); }); }); + +describe('overlay line labels', () => { + const klaud = { + id: 35319969159, + branch: 'klaud/qwen3.5-fp8-gb300-dynamo-sglang-nightly-dev-cu13-20260918-20518d85', + }; + + it('names the hardware and leaves the branch to the legend when one run draws it', () => { + expect(getOverlayLineLabel('GB300 NVL72 (Dynamo SGLang)', klaud, false)).toBe( + '✕ GB300 NVL72 (Dynamo SGLang)', + ); + }); + + it('adds a short run tag only when several runs draw the same hardware', () => { + expect(getOverlayLineLabel('GB200 NVL72 (Dynamo SGLang)', klaud, true)).toBe( + '✕ GB200 NVL72 (Dynamo SGLang) · …20260918-20518d85', + ); + expect(getOverlayLineLabel('B300', { id: 31756025413, branch: 'main' }, true)).toBe( + '✕ B300 · main', + ); + }); + + it.each([ + ['main', 'main'], + ['feat/short-name', 'feat/short-name'], + ['release/2026-09-18-hotfix', '2026-09-18-hotfix'], + [klaud.branch, '…20260918-20518d85'], + ['', 'run 42'], + [null, 'run 42'], + ])('shortens branch %j to %j', (branch, expected) => { + expect(shortRunTag({ id: 42, branch })).toBe(expected); + }); +}); diff --git a/packages/app/src/lib/inference-labels.ts b/packages/app/src/lib/inference-labels.ts index 81f92c485..c9da78643 100644 --- a/packages/app/src/lib/inference-labels.ts +++ b/packages/app/src/lib/inference-labels.ts @@ -7,18 +7,65 @@ interface RunProvenance { run_url?: string | null; } +// Three-week recognition window, ending 2026-10-08 at 21:32 America/New_York. +const UMBP_DSPARK_LABEL_EXPIRES_AT = Date.parse('2026-10-09T01:32:00Z'); + /** Display-only: never change framework/hardware keys used by filters and history. */ export function inferenceFrameworkLabelOverride( framework: string, runUrl?: string | null, ): string | undefined { - return resolveFrameworkAlias(framework) === 'mori-sglang' && - runIdFromRunUrl(runUrl) === '34926284365' - ? 'MoRI UMBP SGLang' - : undefined; + if (resolveFrameworkAlias(framework) !== 'mori-sglang') return undefined; + const runId = runIdFromRunUrl(runUrl); + const hasLabel = + runId === '34926284365' || + (runId === '35166686551' && Date.now() < UMBP_DSPARK_LABEL_EXPIRES_AT); + return hasLabel ? 'MoRI UMBP SGLang' : undefined; } /** Keep unofficial-run identity/markers while making its special engine visible. */ +/** Leads every unofficial-run label, in line labels and legend rows alike. */ +export const OVERLAY_LABEL_MARKER = '✕ '; +const RUN_TAG_MAX = 20; +const RUN_TAG_TAIL = 17; + +export interface OverlayRunIdentity { + id: number | string; + branch?: string | null; +} + +/** + * Short, still recognisable name for a run: the whole branch when it is short, + * else its last path segment, else the branch tail (klaud nightlies end in + * `-`). Falls back to the run id when the branch is unknown. + */ +export function shortRunTag(run: OverlayRunIdentity): string { + const branch = run.branch?.trim() || `run ${run.id}`; + if (branch.length <= RUN_TAG_MAX) return branch; + const segment = branch.slice(branch.lastIndexOf('/') + 1); + if (segment.length > 0 && segment.length <= RUN_TAG_MAX) return segment; + return `…${branch.slice(-RUN_TAG_TAIL)}`; +} + +/** ` · ` appended to an overlay line label when other runs draw the same hardware. */ +export function overlayRunTag(run: OverlayRunIdentity): string { + return ` · ${shortRunTag(run)}`; +} + +/** + * Line-label text for an unofficial-run curve. Pills name the hardware, not the + * branch: branch names run to 70+ characters and the legend already carries + * them. The run tag is added only when several overlay runs draw the same + * hardware, so the pills stay distinguishable. + */ +export function getOverlayLineLabel( + hardwareLabel: string, + run: OverlayRunIdentity, + sharesHardware: boolean, +): string { + return `${OVERLAY_LABEL_MARKER}${hardwareLabel}${sharesHardware ? overlayRunTag(run) : ''}`; +} + export function getInferenceRunLabel( label: string, points: readonly (RunProvenance & { framework?: string })[], diff --git a/packages/app/src/lib/power-basis.test.ts b/packages/app/src/lib/power-basis.test.ts new file mode 100644 index 000000000..9f2bf53f1 --- /dev/null +++ b/packages/app/src/lib/power-basis.test.ts @@ -0,0 +1,444 @@ +import { describe, expect, it } from 'vitest'; + +import type { BenchmarkRow } from '@/lib/api'; +import { rowToAggDataEntry, transformBenchmarkRows } from '@/lib/benchmark-transform'; +import { buildDerivedChartFields, getHardwareKey } from '@/lib/chart-utils'; +import { getGpuSpecs } from '@/lib/constants'; +import { + computePowerBasisFields, + modeledFacilityWattsPerGpu, + POWER_BASES, + POWER_BASIS_FIELDS, + POWER_BASIS_LABELS, + powerBasisNormalization, +} from '@/lib/power-basis'; + +// Qwen3.5 B200 c1, run 34175132645: actual rounded telemetry, eight GPUs +// (same fixture as modeled-system-power.test.ts) plus an output rate. +function row(overrides: Partial = {}): BenchmarkRow { + return { + id: 441192, + model: 'qwen3.5', + hardware: 'b200', + framework: 'sglang', + precision: 'fp8', + spec_method: 'none', + disagg: false, + is_multinode: false, + prefill_tp: 8, + prefill_ep: 1, + prefill_dp_attention: false, + prefill_num_workers: 0, + decode_tp: 8, + decode_ep: 1, + decode_dp_attention: false, + decode_num_workers: 0, + num_prefill_gpu: 8, + num_decode_gpu: 8, + benchmark_type: 'single_turn', + isl: 8192, + osl: 1024, + conc: 1, + offload_mode: 'off', + image: 'lmsysorg/sglang:v0.5.19-cu130', + date: '2026-09-08', + run_url: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34175132645/attempts/1', + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 349.859, + avg_total_gpu_power_w: 2798.868, + total_gpu_energy_j: 120361.299, + joules_per_output_token: 12.937902, + tput_per_gpu: 270, + output_tput_per_gpu: 30, + pp: 1, + pcp_size: 1, + median_intvty: 100, + }, + ...overrides, + }; +} + +/** Full official-path derivation for one row with real HW_REGISTRY specs. */ +function derive(source: BenchmarkRow) { + const entry = rowToAggDataEntry(source); + const hwKey = getHardwareKey(entry); + return { entry, hwKey, fields: buildDerivedChartFields(entry, hwKey) }; +} + +const y = (metric: { y: number } | undefined) => metric?.y; + +describe('computePowerBasisFields', () => { + // H200-like aggregate deployment: 8 GPUs at 50 output tok/s each. + const h200 = { + tdpWatts: 700, + utilityWatts: 1370, + allocatedGpus: 8, + totalOutputTokPerSec: 400, + measuredWatts: 350, + measuredJPerOutputToken: 7, + modeledFacilityWattsPerGpu: 650, + }; + + it('derives provisioned and modeled boundaries for an aggregate deployment', () => { + const values = computePowerBasisFields(h200); + expect(values).toMatchObject({ + gpuProvisionedWatts: 700, + gpuProvisionedJPerOutputToken: 14, + utilityProvisionedWatts: 1370, + utilityModeledWatts: 650, + utilityModeledJPerOutputToken: 13, + }); + expect(values.utilityProvisionedJPerOutputToken).toBeCloseTo(27.4, 10); + }); + + it('charges the prefill pool to every output token of a 4P+4D deployment', () => { + // GB200-like disaggregation: only 4 decode GPUs emit tokens (50 tok/s each), + // but all 8 allocated GPUs are provisioned, so J/token doubles versus a + // per-decode-GPU reading (1200 / 50 = 24). + const values = computePowerBasisFields({ + tdpWatts: 1200, + utilityWatts: 1870, + allocatedGpus: 8, + totalOutputTokPerSec: 200, + measuredWatts: null, + measuredJPerOutputToken: null, + modeledFacilityWattsPerGpu: null, + }); + expect(values.gpuProvisionedJPerOutputToken).toBe(48); + expect(values.utilityProvisionedJPerOutputToken).toBeCloseTo(74.8, 10); + expect(values.utilityModeledWatts).toBeNull(); + expect(values.utilityModeledJPerOutputToken).toBeNull(); + }); + + it('nulls only the boundaries whose inputs are unavailable', () => { + const noModel = computePowerBasisFields({ ...h200, modeledFacilityWattsPerGpu: null }); + expect(noModel.utilityModeledWatts).toBeNull(); + expect(noModel.utilityModeledJPerOutputToken).toBeNull(); + expect(noModel.gpuProvisionedJPerOutputToken).toBe(14); + expect(noModel.utilityProvisionedJPerOutputToken).toBeCloseTo(27.4, 10); + + // B4 is anchored on B1: without measured watts neither modeled value exists, + // even when a caller supplies a modeled W from elsewhere. + const noMeasured = computePowerBasisFields({ + ...h200, + measuredWatts: null, + measuredJPerOutputToken: null, + }); + expect(noMeasured.utilityModeledWatts).toBeNull(); + expect(noMeasured.utilityModeledJPerOutputToken).toBeNull(); + expect(noMeasured.gpuProvisionedWatts).toBe(700); + expect(noMeasured.utilityProvisionedWatts).toBe(1370); + + const noMeasuredEnergy = computePowerBasisFields({ ...h200, measuredJPerOutputToken: null }); + expect(noMeasuredEnergy.utilityModeledWatts).toBe(650); + expect(noMeasuredEnergy.utilityModeledJPerOutputToken).toBeNull(); + + const noThroughput = computePowerBasisFields({ ...h200, totalOutputTokPerSec: null }); + expect(noThroughput.gpuProvisionedWatts).toBe(700); + expect(noThroughput.gpuProvisionedJPerOutputToken).toBeNull(); + expect(noThroughput.utilityProvisionedJPerOutputToken).toBeNull(); + expect(noThroughput.utilityModeledJPerOutputToken).toBe(13); + }); + + it.each([0, -1, NaN, Infinity])('never turns %s into a plotted value', (bad) => { + const values = computePowerBasisFields({ + tdpWatts: bad, + utilityWatts: bad, + allocatedGpus: bad, + totalOutputTokPerSec: bad, + measuredWatts: bad, + measuredJPerOutputToken: bad, + modeledFacilityWattsPerGpu: bad, + }); + expect(Object.values(values).every((v) => v === null)).toBe(true); + }); +}); + +describe('powerBasisNormalization', () => { + it('keeps aggregate rows on the per-GPU ratio because N_alloc cancels', () => { + expect( + powerBasisNormalization({ + output_tput_per_gpu: 50, + disagg: false, + benchmark_type: 'single_turn', + num_prefill_gpu: 8, + num_decode_gpu: 8, + }), + ).toEqual({ allocatedGpus: 1, totalOutputTokPerSec: 50 }); + }); + + it('counts prefill GPUs and decode-only output for fixed-sequence disaggregation', () => { + const disagg = { + output_tput_per_gpu: 50, + disagg: true, + benchmark_type: 'single_turn', + num_prefill_gpu: 4, + num_decode_gpu: 4, + }; + expect(powerBasisNormalization(disagg)).toEqual({ + allocatedGpus: 8, + totalOutputTokPerSec: 200, + }); + expect(powerBasisNormalization({ ...disagg, num_prefill_gpu: 0 })).toEqual({ + allocatedGpus: null, + totalOutputTokPerSec: null, + }); + expect(powerBasisNormalization({ ...disagg, benchmark_type: 'agentic_traces' })).toEqual({ + allocatedGpus: null, + totalOutputTokPerSec: null, + }); + expect(powerBasisNormalization({ ...disagg, output_tput_per_gpu: 0 })).toEqual({ + allocatedGpus: null, + totalOutputTokPerSec: null, + }); + }); +}); + +describe('power boundaries through the derived-field builder', () => { + it('orders B3 ≥ B4 ≥ B1 and B2 ≥ B1 for watts and energy on real B200 telemetry', () => { + const { entry, fields } = derive(row()); + const b1W = y(fields.measuredAvgPower)!; + const b2W = y(fields.gpuProvisionedWatts)!; + const b3W = y(fields.utilityProvisionedWatts)!; + const b4W = y(fields.utilityModeledWatts)!; + expect(b1W).toBe(349.859); + expect(b2W).toBe(getGpuSpecs('b200').tdp); + expect(b3W).toBe(getGpuSpecs('b200').power * 1000); + expect(b4W).toBeCloseTo(6288.4 / 8, 6); + expect(b3W).toBeGreaterThanOrEqual(b4W); + expect(b4W).toBeGreaterThanOrEqual(b1W); + expect(b2W).toBeGreaterThanOrEqual(b1W); + + const b1J = y(fields.measuredJPerOutputToken)!; + const b2J = y(fields.gpuProvisionedJPerOutputToken)!; + const b3J = y(fields.utilityProvisionedJPerOutputToken)!; + const b4J = y(fields.utilityModeledJPerOutputToken)!; + expect(b1J).toBe(12.937902); + expect(b2J).toBeCloseTo(b2W / 30, 10); + expect(b3J).toBeCloseTo(b3W / 30, 10); + expect(b4J).toBeCloseTo((b1J * b4W) / b1W, 10); + expect(b3J).toBeGreaterThanOrEqual(b4J); + expect(b4J).toBeGreaterThanOrEqual(b1J); + expect(b2J).toBeGreaterThanOrEqual(b1J); + + // B4 W ÷ B1 W is the facility-to-board ratio of the attached estimate. + const model = entry.modeledSystemPower; + expect(model?.status).toBe('supported'); + if (model?.status !== 'supported') return; + expect(b4W / b1W).toBeCloseTo(model.deploymentFacilityWatts / (model.gpuCount * b1W), 10); + }); + + it('applies PUE exactly once: B4 W ÷ chassis AC per GPU equals the model PUE', () => { + const { entry, fields } = derive(row()); + const model = entry.modeledSystemPower; + if (model?.status !== 'supported') throw new Error('fixture must be supported'); + const chassisAcPerMeasuredGpu = model.chassisAcWatts / model.gpuCount; + expect(y(fields.utilityModeledWatts)! / chassisAcPerMeasuredGpu).toBeCloseTo(model.pue, 4); + expect(y(fields.utilityModeledWatts)! / y(fields.modeledChassisPowerPerGpu)!).toBeCloseTo( + model.pue, + 4, + ); + }); + + it('normalizes a fixed-sequence 4P+4D row by all eight GPUs while jOutput stays per decode GPU', () => { + const { entry, fields } = derive( + row({ + hardware: 'gb200', + disagg: true, + prefill_tp: 4, + decode_tp: 4, + num_prefill_gpu: 4, + num_decode_gpu: 4, + metrics: { ...row().metrics, tput_per_gpu: 450, output_tput_per_gpu: 50 }, + }), + ); + const specs = getGpuSpecs('gb200'); + expect(y(fields.gpuProvisionedJPerOutputToken)).toBeCloseTo((specs.tdp * 8) / (50 * 4), 10); + expect(y(fields.utilityProvisionedJPerOutputToken)).toBeCloseTo( + (specs.power * 1000 * 8) / (50 * 4), + 10, + ); + // Legacy all-in energy divides one decode GPU's power by that GPU's output. + expect(y(fields.jOutput)).toBeCloseTo((specs.power * 1000) / 50, 10); + expect(y(fields.utilityProvisionedJPerOutputToken)).toBeCloseTo(2 * y(fields.jOutput)!, 10); + // Telemetry is validated (B1 renders) but GB200 NVL72 has no chassis + // model, so the modeled boundary alone is absent. + expect(fields.measuredAvgPower).toBeDefined(); + expect(entry.modeledSystemPower).toMatchObject({ status: 'unsupported', reason: 'hardware' }); + expect(fields.utilityModeledWatts).toBeUndefined(); + expect(fields.utilityModeledJPerOutputToken).toBeUndefined(); + }); + + it('divides B4 by the measured GPU count, not the modeled chassis count, on a 4P+8D H200 deployment', () => { + // Two worker hosts, a half-filled prefill chassis and a full decode chassis: + // the model evaluates 16 GPUs (two chassis) while twelve are measured. + const { entry, fields } = derive( + row({ + hardware: 'h200', + disagg: true, + is_multinode: true, + prefill_tp: 4, + decode_tp: 8, + num_prefill_gpu: 4, + num_decode_gpu: 8, + workers: [ + { + role: 'prefill', + worker_idx: 0, + num_gpus: 4, + hosts: ['prefill-node'], + avg_power_w: 400, + }, + { role: 'decode', worker_idx: 0, num_gpus: 8, hosts: ['decode-node'], avg_power_w: 300 }, + ], + metrics: { + power_valid: 1, + power_metric_schema_version: 2, + avg_power_w: 4000 / 12, + avg_total_gpu_power_w: 4000, + prefill_avg_power_w: 400, + decode_avg_power_w: 300, + joules_per_output_token: 14, + tput_per_gpu: 450, + output_tput_per_gpu: 50, + }, + }), + ); + const model = entry.modeledSystemPower; + expect(model?.status).toBe('supported'); + if (model?.status !== 'supported') return; + expect(model.gpuCount).toBe(12); + expect(model.modeledGpuCount).toBe(16); + expect(model.chassisBasis).toBe('extrapolated'); + + const b1W = y(fields.measuredAvgPower)!; + const b3W = y(fields.utilityProvisionedWatts)!; + const b4W = y(fields.utilityModeledWatts)!; + expect(b1W).toBeCloseTo(4000 / 12, 10); + expect(b4W).toBeCloseTo(model.deploymentFacilityWatts / model.gpuCount, 10); + expect(b4W).not.toBeCloseTo(model.facilityWatts / model.modeledGpuCount, 6); + expect(b4W).not.toBeCloseTo(model.facilityWatts / model.gpuCount, 6); + expect(b4W).not.toBeCloseTo(model.deploymentFacilityWatts / model.modeledGpuCount, 6); + + const b1J = y(fields.measuredJPerOutputToken)!; + const b3J = y(fields.utilityProvisionedJPerOutputToken)!; + const b4J = y(fields.utilityModeledJPerOutputToken)!; + expect(b1J).toBe(14); + expect(b4J).toBeCloseTo((14 * b4W) / (4000 / 12), 10); + expect(b3J).toBeCloseTo(1.5 * y(fields.jOutput)!, 10); + expect(b3W).toBeGreaterThanOrEqual(b4W); + expect(b4W).toBeGreaterThanOrEqual(b1W); + expect(b3J).toBeGreaterThanOrEqual(b4J); + expect(b4J).toBeGreaterThanOrEqual(b1J); + }); + + it('omits the modeled boundary off the 8k/1k workload but keeps provisioned ones', () => { + const { fields } = derive(row({ isl: 1024, osl: 1024 })); + expect(fields.measuredAvgPower).toBeDefined(); + expect(fields.utilityModeledWatts).toBeUndefined(); + expect(fields.utilityModeledJPerOutputToken).toBeUndefined(); + expect(y(fields.gpuProvisionedWatts)).toBe(getGpuSpecs('b200').tdp); + expect(fields.utilityProvisionedJPerOutputToken).toBeDefined(); + }); + + it('omits the modeled boundary with B1 when telemetry is invalid, keeps provisioned ones', () => { + const invalid = derive(row({ metrics: { ...row().metrics, power_valid: 0 } })).fields; + expect(invalid.measuredAvgPower).toBeUndefined(); + expect(invalid.utilityModeledWatts).toBeUndefined(); + expect(invalid.utilityModeledJPerOutputToken).toBeUndefined(); + expect(invalid.gpuProvisionedWatts).toBeDefined(); + expect(invalid.utilityProvisionedWatts).toBeDefined(); + }); + + it('renders the modeled boundary wherever B1 renders, including validated unversioned single-node rows', () => { + const { power_metric_schema_version: _schema, ...unversioned } = row().metrics; + const legacy = derive(row({ metrics: unversioned })); + const model = legacy.entry.modeledSystemPower; + expect(model).toMatchObject({ + status: 'supported', + telemetryBasis: 'validated-unversioned-single-node', + }); + if (model?.status !== 'supported') return; + expect(legacy.fields.measuredAvgPower).toBeDefined(); + expect(legacy.fields.measuredJPerOutputToken).toBeDefined(); + expect( + y(legacy.fields.utilityModeledWatts)! / y(legacy.fields.modeledChassisPowerPerGpu)!, + ).toBeCloseTo(model.pue, 4); + expect(y(legacy.fields.utilityModeledJPerOutputToken)).toBeCloseTo( + (y(legacy.fields.measuredJPerOutputToken)! * y(legacy.fields.utilityModeledWatts)!) / + y(legacy.fields.measuredAvgPower)!, + 10, + ); + // Same numbers as the versioned row: the schema marker changes admission, not the estimate. + expect(legacy.fields.utilityModeledWatts).toEqual(derive(row()).fields.utilityModeledWatts); + }); + + it('omits provisioned boundaries for hardware without registry specs', () => { + const { fields } = derive(row({ hardware: 'unlistedchip' })); + expect(fields.gpuProvisionedWatts).toBeUndefined(); + expect(fields.gpuProvisionedJPerOutputToken).toBeUndefined(); + expect(fields.utilityProvisionedWatts).toBeUndefined(); + expect(fields.utilityProvisionedJPerOutputToken).toBeUndefined(); + }); + + it('serves the same fields to ?unofficialrun= overlays through transformBenchmarkRows', () => { + const { chartData } = transformBenchmarkRows([row()], 'median', 'external'); + const point = chartData[0][0]; + const official = derive(row()).fields; + for (const basis of Object.values(POWER_BASIS_FIELDS)) { + expect(point[basis.watts]).toEqual(official[basis.watts]); + expect(point[basis.energy]).toEqual(official[basis.energy]); + expect(point[basis.watts]?.roof).toBe(false); + } + }); + + it('emits exactly the requested boundary fields for historical trends', () => { + const { entry, hwKey } = derive(row()); + const requested = buildDerivedChartFields(entry, hwKey, [ + 'utilityModeledWatts', + 'gpuProvisionedJPerOutputToken', + ]); + expect(Object.keys(requested).sort()).toEqual( + ['gpuProvisionedJPerOutputToken', 'utilityModeledWatts'].sort(), + ); + const unavailable = buildDerivedChartFields( + rowToAggDataEntry(row({ hardware: 'gb200' })), + 'gb200', + ['tpPerGpu', 'utilityModeledWatts'], + ); + expect(Object.keys(unavailable)).toEqual(['tpPerGpu']); + }); +}); + +describe('modeledFacilityWattsPerGpu', () => { + const supported = rowToAggDataEntry(row()).modeledSystemPower; + + it('divides the deployment facility share by the measured GPU count', () => { + expect(modeledFacilityWattsPerGpu({ modeledSystemPower: supported })).toBeCloseTo( + 6288.4 / 8, + 6, + ); + }); + + it('returns null without a supported model', () => { + expect( + modeledFacilityWattsPerGpu({ + modeledSystemPower: { status: 'unsupported', reason: 'hardware', modelRevision: 'x' }, + }), + ).toBeNull(); + expect(modeledFacilityWattsPerGpu({ modeledSystemPower: undefined })).toBeNull(); + }); +}); + +describe('power basis registry', () => { + it('labels every basis in both locales and maps every derived basis to two fields', () => { + for (const basis of POWER_BASES) { + expect(POWER_BASIS_LABELS[basis].en.length).toBeGreaterThan(0); + expect(POWER_BASIS_LABELS[basis].zh.length).toBeGreaterThan(0); + } + const keys = Object.values(POWER_BASIS_FIELDS).flatMap((f) => [f.watts, f.energy]); + expect(new Set(keys).size).toBe(6); + }); +}); diff --git a/packages/app/src/lib/power-basis.ts b/packages/app/src/lib/power-basis.ts new file mode 100644 index 000000000..034eeca2d --- /dev/null +++ b/packages/app/src/lib/power-basis.ts @@ -0,0 +1,211 @@ +/** + * Power boundaries for one benchmark point, from the GPU board out to the + * utility meter. Pure numbers only — no React, DOM, or registry imports — so + * the derived-field builder, overlays, and tests share one formula set. + * + * | Basis | W / GPU | J / output token | + * | ---------------------- | ------------------------------------------------ | ----------------------------------------- | + * | B1 gpu-measured | `avg_power_w` (existing `measuredAvgPower`) | `joules_per_output_token` (existing) | + * | B2 gpu-provisioned | `HW_REGISTRY.tdp` | W × N_alloc ÷ total output tok/s | + * | B3 utility-provisioned | `HW_REGISTRY.power` × 1000 | W × N_alloc ÷ total output tok/s | + * | B4 utility-modeled | modeled `deploymentFacilityWatts` ÷ `gpuCount` | B1 J/out × (B4 W ÷ B1 W) | + * + * N_alloc counts every allocated GPU (prefill + decode for disaggregation); + * total output tok/s is the whole deployment's. B4 reuses the estimate that + * `modelSystemPower` already attached to the entry — PUE is applied exactly + * once inside that model, never here — and is withheld wherever B1 is. + * Unavailable values are `null`; callers omit the field rather than plotting 0. + */ +import type { AggDataEntry, InferenceData, PowerBasisFieldKey } from '@/components/inference/types'; + +export const POWER_BASES = [ + 'gpu-measured', + 'gpu-provisioned', + 'utility-provisioned', + 'utility-modeled', +] as const; +export type PowerBasis = (typeof POWER_BASES)[number]; +export type PowerQuantity = 'watts' | 'energy'; + +export const POWER_BASIS_LABELS: Record = { + 'gpu-measured': { en: 'GPU measured', zh: 'GPU 实测' }, + 'gpu-provisioned': { en: 'GPU provisioned (TDP)', zh: 'GPU 额定(TDP)' }, + 'utility-provisioned': { en: 'Utility provisioned (all-in)', zh: '全电源配置(all-in)' }, + 'utility-modeled': { en: 'Utility modeled (PUE)', zh: '数据中心建模(含 PUE)' }, +}; + +/** InferenceData keys per derived basis and quantity. B1 lives on the measured* fields. */ +export const POWER_BASIS_FIELDS: Record< + Exclude, + Record +> = { + 'gpu-provisioned': { + watts: 'gpuProvisionedWatts', + energy: 'gpuProvisionedJPerOutputToken', + }, + 'utility-provisioned': { + watts: 'utilityProvisionedWatts', + energy: 'utilityProvisionedJPerOutputToken', + }, + 'utility-modeled': { + watts: 'utilityModeledWatts', + energy: 'utilityModeledJPerOutputToken', + }, +}; + +export interface PowerBasisInput { + /** B2 W/GPU: HW_REGISTRY tdp. 0 means the spec is not yet available. */ + tdpWatts: number | null; + /** B3 W/GPU: HW_REGISTRY all-in power, already in watts. */ + utilityWatts: number | null; + /** Every GPU the deployment occupies (prefill + decode for disaggregation). */ + allocatedGpus: number | null; + /** Whole-deployment successful output tokens per second. */ + totalOutputTokPerSec: number | null; + /** B1 W/GPU from validated telemetry. */ + measuredWatts: number | null; + /** B1 J/output token from the same telemetry window. */ + measuredJPerOutputToken: number | null; + /** + * B4 W/GPU: modeled facility watts (PUE already applied) per measured GPU. + * B4 is a scaling of B1, so it is withheld whenever `measuredWatts` is null. + */ + modeledFacilityWattsPerGpu: number | null; +} + +export type PowerBasisValues = Record; + +const positive = (value: unknown): value is number => + typeof value === 'number' && Number.isFinite(value) && value > 0; +const count = (value: unknown): value is number => positive(value) && Number.isSafeInteger(value); +const orNull = (value: number): number | null => (positive(value) ? value : null); + +/** + * Derives the B2–B4 boundary values from plain numbers. Any unavailable input + * yields `null` for the values that depend on it and leaves the rest intact. + */ +export function computePowerBasisFields(input: PowerBasisInput): PowerBasisValues { + const tdp = positive(input.tdpWatts) ? input.tdpWatts : null; + const utility = positive(input.utilityWatts) ? input.utilityWatts : null; + const measuredWatts = positive(input.measuredWatts) ? input.measuredWatts : null; + const measuredJ = positive(input.measuredJPerOutputToken) ? input.measuredJPerOutputToken : null; + // B4 is B1 carried out to the utility meter, so it follows B1's availability: + // no measured watts, no modeled boundary (B3 ≥ B4 ≥ B1 needs its anchor). + const modeled = + measuredWatts !== null && positive(input.modeledFacilityWattsPerGpu) + ? input.modeledFacilityWattsPerGpu + : null; + + // Provisioned energy: GPU-seconds spent per output token by the whole + // deployment (N_alloc ÷ total tok/s) × W per GPU = J per output token. + const gpuSecondsPerOutputToken = + positive(input.allocatedGpus) && positive(input.totalOutputTokPerSec) + ? input.allocatedGpus / input.totalOutputTokPerSec + : null; + const provisionedEnergy = (watts: number | null) => + watts !== null && gpuSecondsPerOutputToken !== null + ? orNull(watts * gpuSecondsPerOutputToken) + : null; + + // Modeled energy scales the producer's same-window E/N by modeled ÷ measured W, + // so it inherits B1's token denominator instead of re-deriving one. + const modeledEnergy = + modeled !== null && measuredWatts !== null && measuredJ !== null + ? orNull((measuredJ * modeled) / measuredWatts) + : null; + + return { + gpuProvisionedWatts: tdp, + gpuProvisionedJPerOutputToken: provisionedEnergy(tdp), + utilityProvisionedWatts: utility, + utilityProvisionedJPerOutputToken: provisionedEnergy(utility), + utilityModeledWatts: modeled, + utilityModeledJPerOutputToken: modeledEnergy, + }; +} + +type PowerBasisEntry = Pick< + AggDataEntry, + | 'output_tput_per_gpu' + | 'disagg' + | 'benchmark_type' + | 'num_prefill_gpu' + | 'num_decode_gpu' + | 'avg_power_w' + | 'joules_per_output_token' + | 'modeledSystemPower' +>; + +/** + * Whole-deployment normalization for the provisioned energies. Aggregate rows + * already report output per allocated GPU, so N_alloc cancels and the ratio + * 1 GPU : per-GPU throughput is exact without trusting display counts (legacy + * ingest can encode TP × EP twice). Fixed-sequence disaggregated rows report + * output per decode GPU while the deployment also powers the prefill pool, so + * total output = per-GPU × decode GPUs and N_alloc = prefill + decode GPUs. + * Other disaggregated benchmark types are left out: whether AgentX throughput + * already divides by all GPUs is not verifiable in-app. + */ +export function powerBasisNormalization( + entry: Pick< + PowerBasisEntry, + 'output_tput_per_gpu' | 'disagg' | 'benchmark_type' | 'num_prefill_gpu' | 'num_decode_gpu' + >, +): Pick { + const perGpu = entry.output_tput_per_gpu; + const unavailable = { allocatedGpus: null, totalOutputTokPerSec: null }; + if (!positive(perGpu)) return unavailable; + if (!entry.disagg) return { allocatedGpus: 1, totalOutputTokPerSec: perGpu }; + if (entry.benchmark_type !== 'single_turn') return unavailable; + const prefill = entry.num_prefill_gpu; + const decode = entry.num_decode_gpu; + if (!count(prefill) || !count(decode)) return unavailable; + return { allocatedGpus: prefill + decode, totalOutputTokPerSec: perGpu * decode }; +} + +/** + * B4 W/GPU from the estimate `rowToAggDataEntry` attached. The model owns + * telemetry admission: `modelSystemPower` requires `power_valid === 1` plus + * schema v2, or the validated unversioned single-node producer it records as + * `telemetryBasis: 'validated-unversioned-single-node'`. That is the same + * population the app plots as B1 (`measuredAvgPower`) and as + * `modeledChassisPowerPerGpu`, so B4 renders exactly where they do. The public + * API's stricter `strictV2` row filter is not re-applied here; it is not + * applied to the chart's B1 either. + */ +export function modeledFacilityWattsPerGpu( + entry: Pick, +): number | null { + const model = entry.modeledSystemPower; + if (model?.status !== 'supported') return null; + if (!positive(model.deploymentFacilityWatts) || !count(model.gpuCount)) return null; + return orNull(model.deploymentFacilityWatts / model.gpuCount); +} + +export type PowerBasisChartFields = Partial>; + +/** + * Chart-shaped B2–B4 fields for one entry. Keys are present only for finite, + * positive values: the metric filters drop a point by `metricKey in point`, + * and the coordinate remap falls back to raw throughput when a key exists + * with an unusable value. + */ +export function buildPowerBasisChartFields( + entry: PowerBasisEntry, + specs: { tdp?: number; power?: number }, +): PowerBasisChartFields { + const values = computePowerBasisFields({ + tdpWatts: specs.tdp ?? null, + utilityWatts: positive(specs.power) ? specs.power * 1000 : null, + ...powerBasisNormalization(entry), + measuredWatts: entry.avg_power_w ?? null, + measuredJPerOutputToken: entry.joules_per_output_token ?? null, + modeledFacilityWattsPerGpu: modeledFacilityWattsPerGpu(entry), + }); + const fields: PowerBasisChartFields = {}; + for (const key of Object.keys(values) as PowerBasisFieldKey[]) { + const y = values[key]; + if (y !== null) fields[key] = { y, roof: false }; + } + return fields; +} diff --git a/packages/app/src/lib/rankings.test.ts b/packages/app/src/lib/rankings.test.ts index c42921555..149450aa2 100644 --- a/packages/app/src/lib/rankings.test.ts +++ b/packages/app/src/lib/rankings.test.ts @@ -67,7 +67,7 @@ describe('rankings registry', () => { it('has one page per (kind, model) pair', () => { const entries = getAllRankingPageEntries(); expect(entries.length).toBe(RANKING_KINDS.length * INFERENCE_MODEL_SLUGS.length); - expect(entries.length).toBe(26); + expect(entries.length).toBe(30); }); it('has unique, well-formed slugs', () => { diff --git a/packages/app/src/lib/run-pages.test.ts b/packages/app/src/lib/run-pages.test.ts index bb34f73f8..5e9368a9c 100644 --- a/packages/app/src/lib/run-pages.test.ts +++ b/packages/app/src/lib/run-pages.test.ts @@ -35,7 +35,7 @@ describe('run pages registry', () => { it('has one candidate per (model, chip) pair', () => { const entries = getAllRunPageEntries(); expect(entries.length).toBe(INFERENCE_MODEL_SLUGS.length * getAllChipPages().length); - expect(entries.length).toBe(117); + expect(entries.length).toBe(135); }); it('has unique, well-formed slugs', () => { diff --git a/packages/app/src/lib/tab-meta-zh.ts b/packages/app/src/lib/tab-meta-zh.ts index d52484d66..4b580fbb5 100644 --- a/packages/app/src/lib/tab-meta-zh.ts +++ b/packages/app/src/lib/tab-meta-zh.ts @@ -42,6 +42,16 @@ export const TAB_META_ZH: Record = { '本页面提供吞吐量与总拥有成本(TCO)计算器:基于真实基准测试数据,估算不同芯片配置下 LLM 推理服务的每百万 token 成本与性价比。', fleet: '本页面提供集群生命周期经济性分析:按设施功率预算确定固定集群的规模,从模型发布之日起,基于历史基准测试中实测的软件配置改进,测算收入、成本、利润与回本时间。', + 'first-token': + '本页面展示首 token 延迟约束下的成本对比:在选定的最低交互性下,按 2 s、5 s、10 s 等各档首 token 延迟(TTFT)上限,逐档找出各芯片厂商每百万 token 成本最低的实测配置,并标注最优配置与厂商间的成本差距。所有柱形均来自实际运行过的配置,不做插值,可直接追溯到对应的 GitHub Actions 运行记录。', + 'cache-reuse': + '本页面展示前缀缓存复用情况:选定模型与配置后,按并发数逐档以堆叠柱形拆分 prompt token 的来源,分为芯片 HBM 缓存命中、主机层缓存命中与未复用(重新计算)三部分,并可叠加 trace 的理论上限。所有数值均来自运行时实测的缓存计数,可追溯到对应的 GitHub Actions 运行记录。', 'profit-estimator': '本页面提供推理利润估算器:在选定的交互性与利用率下,按每芯片每小时计算各芯片的收入,并以堆叠柱形拆分为算力支出(TCO $/chip/hr)、模型实验室从收入中抽取的许可费,以及运营方所剩利润。', 'profit-estimator-per-gigawatt': @@ -148,6 +162,8 @@ export const TAB_LABELS_ZH: Record = { historical: '历史趋势', calculator: 'TCO 计算器', fleet: '集群生命周期', + 'first-token': '首 token 延迟约束', + 'cache-reuse': '前缀缓存复用', 'profit-estimator': '利润估算', 'profit-estimator-per-gigawatt': '每吉瓦利润估算', reliability: '可靠性', diff --git a/packages/app/src/lib/tab-meta.ts b/packages/app/src/lib/tab-meta.ts index 95c368314..d1aa2e096 100644 --- a/packages/app/src/lib/tab-meta.ts +++ b/packages/app/src/lib/tab-meta.ts @@ -46,6 +46,16 @@ export const TAB_META: Record { expect(PARAM_DEFAULTS.i_revenue).toBe('normalized'); }); + it('has an empty default for perf rulers (none placed)', async () => { + const { PARAM_DEFAULTS } = await import('@/lib/url-state'); + expect(PARAM_DEFAULTS.i_rulers).toBe(''); + }); + + it('has an empty default for the power comparison (metric drawn alone)', async () => { + const { PARAM_DEFAULTS } = await import('@/lib/url-state'); + expect(PARAM_DEFAULTS.i_pcompare).toBe(''); + }); + it('has empty string defaults for legend-active params', async () => { const { PARAM_DEFAULTS } = await import('@/lib/url-state'); expect(PARAM_DEFAULTS.i_active).toBe(''); @@ -141,6 +151,17 @@ describe('readUrlParams', () => { expect(params.i_seq).toBe('2k/4k'); }); + it('reads perf rulers decoded, with curve ids and separators intact', async () => { + const encoded = new URLSearchParams({ + i_rulers: '41.5|roofline-b200_trt_fp8|overlay-roofline-h100_vllm_fp8_run1__2026-09%2F11', + }).toString(); + setupWindow(`?${encoded}`); + const { readUrlParams } = await import('@/lib/url-state'); + expect(readUrlParams().i_rulers).toBe( + '41.5|roofline-b200_trt_fp8|overlay-roofline-h100_vllm_fp8_run1__2026-09%2F11', + ); + }); + it('reads the Internal TCO basis from the URL', async () => { setupWindow('?g_tco=internal'); const { readUrlParams } = await import('@/lib/url-state'); @@ -370,6 +391,17 @@ describe('writeUrlParams + buildShareUrl', () => { expect(url).not.toContain('g_model'); }); + it('drops i_rulers from the share link once the last ruler is cleared', async () => { + setupWindow('?i_rulers=41.5%7Croofline-a%7Croofline-b', '/inference'); + const { writeUrlParams, buildShareUrl } = await import('@/lib/url-state'); + expect(buildShareUrl()).toContain('i_rulers='); + + writeUrlParams({ i_rulers: '' }); + await vi.advanceTimersByTimeAsync(200); + + expect(buildShareUrl()).not.toContain('i_rulers'); + }); + it('removes the revenue source instead of emitting an empty query param off-metric', async () => { setupWindow('?i_revenue=openrouter', '/inference'); const { writeUrlParams, buildShareUrl } = await import('@/lib/url-state'); @@ -639,6 +671,28 @@ describe('buildShareUrl tab filtering', () => { expect(url).not.toContain('r_active'); }); + it('shares i_rulers on /inference and its compare routes but not on other tabs', async () => { + const rulers = '41.5|roofline-b200_trt_fp8|roofline-mi355x_sglang_fp4'; + for (const pathname of ['/inference', '/zh/inference', '/compare/b200-vs-mi355x']) { + vi.resetModules(); + setupWindow('', pathname); + const { writeUrlParams, buildShareUrl } = await import('@/lib/url-state'); + writeUrlParams({ i_rulers: rulers }); + await vi.advanceTimersByTimeAsync(200); + const url = buildShareUrl(); + expect(url, pathname).toContain('i_rulers='); + expect(new URL(url).searchParams.get('i_rulers'), pathname).toBe(rulers); + } + for (const pathname of ['/evaluation', '/reliability']) { + vi.resetModules(); + setupWindow('', pathname); + const { writeUrlParams, buildShareUrl } = await import('@/lib/url-state'); + writeUrlParams({ i_rulers: rulers }); + await vi.advanceTimersByTimeAsync(200); + expect(buildShareUrl(), pathname).not.toContain('i_rulers'); + } + }); + it('includes e_active on /evaluation but not i_active or r_active', async () => { setupWindow('', '/evaluation'); const { writeUrlParams, buildShareUrl } = await import('@/lib/url-state'); @@ -989,3 +1043,37 @@ describe('rememberChartStateInUrl — params this module does not own', () => { expect(params.has('unofficialrun')).toBe(false); }); }); + +describe('chartStateHref', () => { + beforeEach(() => { + vi.useFakeTimers(); + vi.resetModules(); + }); + + afterEach(() => { + vi.useRealTimers(); + vi.unstubAllGlobals(); + }); + + it('rebuilds the page URL from the chart state, canonical unofficial-run key and overrides', async () => { + setupWindow( + '?unofficialrun=31415926535&i_metric=y_tpPerGpu&utm_source=x', + '/inference', + '#chart', + ); + const { chartStateHref, writeUrlParams } = await import('@/lib/url-state'); + // A default value (i_prec) is not chart state and stays out of the link. + writeUrlParams({ g_model: 'Qwen-3.5-397B-A17B', i_metric: 'y_measuredAvgPower', i_prec: '' }); + + const url = new URL(chartStateHref({ i_metric: 'y_measuredPowerTimeline' })); + expect(url.origin).toBe('https://example.com'); + expect(url.pathname).toBe('/inference'); + expect(url.hash).toBe('#chart'); + expect(url.searchParams.get('utm_source')).toBe('x'); + expect(url.searchParams.get('g_model')).toBe('Qwen-3.5-397B-A17B'); + expect(url.searchParams.has('i_prec')).toBe(false); + expect(url.searchParams.get('i_metric')).toBe('y_measuredPowerTimeline'); + expect(url.searchParams.get('unofficialruns')).toBe('31415926535'); + expect(url.searchParams.has('unofficialrun')).toBe(false); + }); +}); diff --git a/packages/app/src/lib/url-state.ts b/packages/app/src/lib/url-state.ts index 1ebd9f9a4..aa13a5367 100644 --- a/packages/app/src/lib/url-state.ts +++ b/packages/app/src/lib/url-state.ts @@ -63,6 +63,12 @@ const URL_STATE_KEYS = [ 'i_spec', // Measured-power certification tiers ('certified' / 'legacy', comma-joined). 'i_power', + // Completed Perf Rulers on the primary inference chart: `isoX|curveA|curveB` + // entries joined by `;` (see serializePerfRulers in d3-chart/layers/perf-ruler). + 'i_rulers', + // Comparison series overlaid on a gated power metric: `boundaries` (every + // power boundary) or `roles` (prefill / decode pools). Empty = the metric alone. + 'i_pcompare', // Exact serving-envelope pair behind an Overview 30-day comparison cell. 'i_overview_current', 'i_overview_baseline', @@ -98,6 +104,13 @@ const URL_STATE_KEYS = [ 'c_rec', 'c_life', 'c_power', + // First-token limits: interactivity floor (tok/s/user) and the comma-joined + // TTFT cap ladder in seconds. Empty means the page's sequence-aware defaults. + 'c_ivmin', + 'c_ttft', + // Cache reuse: the configuration group plotted. Empty means the group with + // the most rows reporting cache tiers. + 'c_cfg', ] as const; export type UrlStateKey = (typeof URL_STATE_KEYS)[number]; @@ -171,6 +184,8 @@ export const PARAM_DEFAULTS: Record = { i_disagg: '', i_spec: '', i_power: '', + i_rulers: '', + i_pcompare: '', i_overview_current: '', i_overview_baseline: '', e_rundate: '', @@ -194,6 +209,9 @@ export const PARAM_DEFAULTS: Record = { c_oprice: '', c_life: '', c_power: 'provisioned', + c_ivmin: '', + c_ttft: '', + c_cfg: '', // Empty means the default y metric (margin). c_ly: '', c_ramp: DEFAULT_LIFECYCLE_RAMP_MONTHS, @@ -483,6 +501,26 @@ export function rememberChartStateInUrl(): string { return chartParams.toString(); } +/** + * The current page's URL carrying its chart state plus `overrides`, + * canonicalised like `rememberChartStateInUrl`: chart params and both + * unofficial-run spellings are dropped from the live address bar before the + * store's state (and the overrides) are layered on. For anchors that must + * work with open-in-new-tab, where the in-memory state would otherwise be lost. + */ +export function chartStateHref(overrides: Record): string { + const { origin, pathname, hash, search } = window.location; + const merged = new URLSearchParams(search); + for (const key of URL_STATE_KEYS) merged.delete(key); + // Collected first: deleting while iterating the params would skip entries. + const staleRunKeys = [...merged.keys()].filter((key) => UNOFFICIAL_RUN_PARAM_RE.test(key)); + for (const key of staleRunKeys) merged.delete(key); + for (const [key, value] of collectTabParams()) merged.set(key, value); + for (const [key, value] of Object.entries(overrides)) merged.set(key, value); + const query = merged.toString(); + return `${origin}${pathname}${query ? `?${query}` : ''}${hash}`; +} + /** * Append the current chart state to an outbound in-app href, so the page it * opens can link back to the chart the user left. Used for the agentic diff --git a/packages/app/src/lib/video-alias-redirects.test.ts b/packages/app/src/lib/video-alias-redirects.test.ts new file mode 100644 index 000000000..75b39818f --- /dev/null +++ b/packages/app/src/lib/video-alias-redirects.test.ts @@ -0,0 +1,34 @@ +import { describe, expect, it } from 'vitest'; + +import { VIDEO_ALIAS_REDIRECTS, VIDEO_ROUTE_ALIASES } from './video-alias-redirects'; + +describe('VIDEO_ALIAS_REDIRECTS', () => { + it('redirects every alias to /video in both locale trees', () => { + expect(VIDEO_ALIAS_REDIRECTS).toHaveLength(VIDEO_ROUTE_ALIASES.length * 2); + + for (const alias of VIDEO_ROUTE_ALIASES) { + for (const prefix of ['', '/zh']) { + expect(VIDEO_ALIAS_REDIRECTS).toContainEqual({ + source: `${prefix}/${alias}/:path*`, + destination: `${prefix}/video/:path*`, + permanent: true, + }); + } + } + }); + + it('covers the /xvideo and /xxvideo vanity paths', () => { + const sources = VIDEO_ALIAS_REDIRECTS.map((redirect) => redirect.source); + expect(sources).toContain('/xvideo/:path*'); + expect(sources).toContain('/xxvideo/:path*'); + expect(sources).toContain('/zh/xvideo/:path*'); + expect(sources).toContain('/zh/xxvideo/:path*'); + }); + + it('never redirects the canonical video path to itself', () => { + for (const redirect of VIDEO_ALIAS_REDIRECTS) { + expect(redirect.source).not.toBe(redirect.destination); + expect(redirect.source).not.toMatch(/^(?:\/zh)?\/video\//); + } + }); +}); diff --git a/packages/app/src/lib/video-alias-redirects.ts b/packages/app/src/lib/video-alias-redirects.ts new file mode 100644 index 000000000..d1203349d --- /dev/null +++ b/packages/app/src/lib/video-alias-redirects.ts @@ -0,0 +1,25 @@ +interface VideoAliasRedirect { + source: string; + destination: string; + permanent: true; +} + +/** + * Vanity aliases for the VideoGenX dashboard. `/video` stays the canonical + * home in both locale trees; the aliases 308-redirect there like the other + * alias tables in `next.config.ts`, so shared links keep working and search + * engines see one indexable URL. Query strings are carried automatically by + * Next's redirect handling. + */ +export const VIDEO_ROUTE_ALIASES = ['xvideo', 'xxvideo'] as const; + +const LOCALE_PREFIXES = ['', '/zh'] as const; + +export const VIDEO_ALIAS_REDIRECTS: readonly VideoAliasRedirect[] = LOCALE_PREFIXES.flatMap( + (prefix) => + VIDEO_ROUTE_ALIASES.map((alias) => ({ + source: `${prefix}/${alias}/:path*`, + destination: `${prefix}/video/:path*`, + permanent: true as const, + })), +); diff --git a/packages/app/timings.json b/packages/app/timings.json index 98d6e58d1..2f1e8638d 100644 --- a/packages/app/timings.json +++ b/packages/app/timings.json @@ -44,6 +44,10 @@ "spec": "cypress/e2e/blog.cy.ts", "duration": 5253 }, + { + "spec": "cypress/e2e/cache-reuse.cy.ts", + "duration": 4000 + }, { "spec": "cypress/e2e/calculator-overlay.cy.ts", "duration": 8785 @@ -128,6 +132,10 @@ "spec": "cypress/e2e/evaluation-chart.cy.ts", "duration": 12735 }, + { + "spec": "cypress/e2e/first-token-limits.cy.ts", + "duration": 4000 + }, { "spec": "cypress/e2e/fixed-sequence-logs.cy.ts", "duration": 3002 @@ -228,6 +236,18 @@ "spec": "cypress/e2e/performance.cy.ts", "duration": 2377 }, + { + "spec": "cypress/e2e/powerx-basis.cy.ts", + "duration": 3000 + }, + { + "spec": "cypress/e2e/powerx-timeline.cy.ts", + "duration": 24000 + }, + { + "spec": "cypress/e2e/powerx-compare.cy.ts", + "duration": 24000 + }, { "spec": "cypress/e2e/profit-estimator.cy.ts", "duration": 39450 diff --git a/packages/constants/src/models.ts b/packages/constants/src/models.ts index 47fa5a7d5..10e159ed9 100644 --- a/packages/constants/src/models.ts +++ b/packages/constants/src/models.ts @@ -14,6 +14,13 @@ export const DB_MODEL_TO_DISPLAY: Record = { // Qwen4-architecture preview, not a Qwen3.5 point release (GatedDeltaNet plus // Qwen Sparse Attention, 512 experts), so it gets its own display bucket. 'qwen3.8next': 'Qwen3.8-Flash-Next', + // Qwen3.8-27B is the dense 27B member of the Qwen3.8 family (hybrid GDN linear + // attention, 48 of 64 layers), served bf16 with the RadixArk DSpark drafter. + // `qwen3.827beager` is the same checkpoint served with CUDA graphs disabled + // (--enforce-eager on target and drafter); it keeps its own DB bucket and + // display name so the two serving modes never collapse onto one chart point. + 'qwen3.827b': 'Qwen3.8-27B', + 'qwen3.827beager': 'Qwen3.8-27B-Eager', 'kimik2.5': 'Kimi-K2.5', 'kimik2.6': 'Kimi-K2.5', 'kimik2.7-code': 'Kimi-K2.5', @@ -158,6 +165,11 @@ export const MODEL_RELEASE_DATES: Record = { // precedes the first sweep on 08-27. The model card itself states no date. // sweep: 2026-08-27 — day zero. 'Qwen3.8-Flash-Next': '2026-08-26', + // Apache-2.0 weights at Qwen/Qwen3.8-27B: the Hugging Face repo was created + // 2026-08-05T08:22Z (its first commit) and last modified 2026-08-14. The eager + // bucket is the same checkpoint. sweep: 2026-09-18 (InferenceX#3260). + 'Qwen3.8-27B': '2026-08-05', + 'Qwen3.8-27B-Eager': '2026-08-05', // Bucket covers M2.5 and M2.7, so the date is M2.5's: announced 2026-02-12 // with weights on Hugging Face, architecturally unchanged from M2 (230B/10B). // Was 2025-10-25, which is M2's launch, not M2.5's — `model-architectures.ts` diff --git a/packages/db/src/etl/normalizers.ts b/packages/db/src/etl/normalizers.ts index b8c20736e..437e8aa5f 100644 --- a/packages/db/src/etl/normalizers.ts +++ b/packages/db/src/etl/normalizers.ts @@ -102,6 +102,12 @@ export const MODEL_TO_KEY: Record = { // PREFIX_ALIASES entry is needed. 'Qwen/Qwen3.8-Flash-Next-FP8': 'qwen3.8next', 'RadixArk/Qwen3.8-Flash-Next-NVFP4': 'qwen3.8next', + // Qwen3.8-27B (dense 27B, bf16). Both recipes serve this checkpoint and report + // `infmax_model_prefix` equal to their DB key (`qwen3.827b`, or `qwen3.827beager` + // for the --enforce-eager variant), so the prefix resolves first and this path + // entry only backs rows that lack the prefix. Seen in run 35362390175 + // (H200, InferenceX#3263) and run 35359530151 (H100, InferenceX#3260). + 'Qwen/Qwen3.8-27B': 'qwen3.827b', // Kimi-K2.5 / K2.6 / K2.7-Code (same architecture, distinct DB buckets) 'moonshotai/Kimi-K2.5': 'kimik2.5', 'moonshotai/Kimi-K2.6': 'kimik2.6', From 598a8bef4a8ff5ee4db38b79701e8afc8756c47f Mon Sep 17 00:00:00 2001 From: Wenyao Gao <105094497+edwingao28@users.noreply.github.com> Date: Sun, 20 Sep 2026 20:49:32 -0700 Subject: [PATCH 014/103] =?UTF-8?q?[PowerX]=20validate=20power=20evidence?= =?UTF-8?q?=20and=20preserve=20published=20curves=20/=20=E6=A0=A1=E9=AA=8C?= =?UTF-8?q?=E5=8A=9F=E8=80=97=E8=AF=81=E6=8D=AE=E5=B9=B6=E4=BF=9D=E6=8A=A4?= =?UTF-8?q?=E5=B7=B2=E5=8F=91=E5=B8=83=E6=9B=B2=E7=BA=BF=20(#1152)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat: enforce required PowerX publication coverage 中文:强制校验 PowerX 必需功耗点的入库与发布完整性。 按实际入库身份匹配并发点,保留过滤后的缺失检查,并核对 AgentX 数据库与 API 结果。 * fix: validate power evidence and preserve published curves 中文:校验必需功耗证据,并在入库前保护已发布曲线。版本化 manifest 绑定来源、物理 GPU、测量窗口和产物哈希;有意替换必须精确声明旧快照及移除点。 --- .github/workflows/ingest-agentic-results.yml | 12 + .github/workflows/ingest-results.yml | 2 + docs/data-pipeline.md | 47 +- docs/fixtures/powerx-manifest-v2/README.md | 101 ++++ .../artifacts/agentic_golden/gpu_metrics.csv | 3 + .../agentic_golden/gpu_metrics_identity.csv | 2 + .../artifacts/agentic_golden/power_node.txt | 1 + .../agentic_golden/power_validation.json | 21 + .../artifacts/bmk_agentic_golden/agg.json | 25 + .../changelog_metadata.json | 4 + .../sweep_manifest.json | 98 +++ packages/db/src/etl/power-publication.test.ts | 8 +- packages/db/src/etl/power-publication.ts | 36 +- .../src/etl/required-power-curve-db.test.ts | 133 ++++ .../db/src/etl/required-power-curve.test.ts | 123 ++++ packages/db/src/etl/required-power-curve.ts | 287 +++++++++ .../etl/required-power-publication.test.ts | 572 +++++++++--------- .../db/src/etl/required-power-publication.ts | 520 +++++++++++++++- packages/db/src/ingest-ci-run.ts | 27 +- .../db/src/preflight-power-publication.ts | 36 ++ packages/db/src/prepare-ci-artifacts.test.ts | 90 +++ packages/db/src/prepare-ci-artifacts.ts | 11 + packages/db/src/verify-power-publication.ts | 2 +- 23 files changed, 1837 insertions(+), 324 deletions(-) create mode 100644 docs/fixtures/powerx-manifest-v2/README.md create mode 100644 docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics.csv create mode 100644 docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics_identity.csv create mode 100644 docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_node.txt create mode 100644 docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_validation.json create mode 100644 docs/fixtures/powerx-manifest-v2/artifacts/bmk_agentic_golden/agg.json create mode 100644 docs/fixtures/powerx-manifest-v2/artifacts/changelog-metadata/changelog_metadata.json create mode 100644 docs/fixtures/powerx-manifest-v2/artifacts/required-power-sweep-manifest/sweep_manifest.json create mode 100644 packages/db/src/etl/required-power-curve-db.test.ts create mode 100644 packages/db/src/etl/required-power-curve.test.ts create mode 100644 packages/db/src/etl/required-power-curve.ts create mode 100644 packages/db/src/preflight-power-publication.ts create mode 100644 packages/db/src/prepare-ci-artifacts.test.ts diff --git a/.github/workflows/ingest-agentic-results.yml b/.github/workflows/ingest-agentic-results.yml index b81cd68b0..3dcd23773 100644 --- a/.github/workflows/ingest-agentic-results.yml +++ b/.github/workflows/ingest-agentic-results.yml @@ -31,6 +31,11 @@ on: description: InferenceX Actions run ID to ingest required: true type: string + require-power: + description: Require the versioned PowerX manifest and all declared evidence + required: false + default: false + type: boolean run-attempt: description: InferenceX Actions run attempt to ingest required: false @@ -85,6 +90,11 @@ on: description: InferenceX Actions run ID to ingest required: true type: string + require-power: + description: Require the versioned PowerX manifest and all declared evidence + required: false + default: false + type: boolean run-attempt: description: InferenceX Actions run attempt to ingest required: false @@ -276,6 +286,7 @@ jobs: github.event.client_payload.run-id || inputs.run-id }} ARTIFACTS_PATH: ${{ github.workspace }}/artifacts INGEST_REPO: SemiAnalysisAI/InferenceX + INGEST_REQUIRE_POWER: ${{ github.event.client_payload.require-power == true || inputs.require-power == true }} run: bun run admin:db:prepare:ci - name: Run migrations @@ -288,6 +299,7 @@ jobs: INGEST_RUN_ATTEMPT: ${{ steps.artifacts.outputs.merge-run-attempt }} INGEST_ARTIFACTS_PATH: ${{ github.workspace }}/artifacts INGEST_REPO: SemiAnalysisAI/InferenceX + INGEST_REQUIRE_POWER: ${{ github.event.client_payload.require-power == true || inputs.require-power == true }} UNMAPPED_ENTITIES_OUTPUT: ${{ github.workspace }}/unmapped-entities.json POWER_PUBLICATION_MANIFEST: ${{ github.workspace }}/power-publication.json run: bun run admin:db:ingest:ci diff --git a/.github/workflows/ingest-results.yml b/.github/workflows/ingest-results.yml index 04b253e34..d66e50a8f 100644 --- a/.github/workflows/ingest-results.yml +++ b/.github/workflows/ingest-results.yml @@ -60,6 +60,7 @@ jobs: github.event.client_payload.run-id }} ARTIFACTS_PATH: ${{ github.workspace }}/artifacts INGEST_REPO: SemiAnalysisAI/InferenceX + INGEST_REQUIRE_POWER: ${{ github.event.client_payload.require-power == true }} run: bun run admin:db:prepare:ci - name: Run migrations @@ -75,6 +76,7 @@ jobs: INGEST_RUN_ATTEMPT: ${{ steps.artifacts.outputs.merge-run-attempt }} INGEST_ARTIFACTS_PATH: ${{ github.workspace }}/artifacts INGEST_REPO: SemiAnalysisAI/InferenceX + INGEST_REQUIRE_POWER: ${{ github.event.client_payload.require-power == true }} POWER_PUBLICATION_MANIFEST: ${{ github.workspace }}/power-publication.json UNMAPPED_ENTITIES_OUTPUT: ${{ github.workspace }}/unmapped-entities.json run: bun run admin:db:ingest:ci diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index cf005cd11..65dfb67c6 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -84,23 +84,36 @@ must use the append-only contract below. ### Required Power Publication -Ordinary sweeps that opt into `require-power` upload the producer's -`required-power-sweep-manifest/sweep_manifest.json`. Before any CI ingest upsert, -the app matches its required benchmark rows by recipe fingerprint, concurrency, -and scenario/sequence lengths, then requires valid v2 power and positive energy. -Disaggregated recipes also require both role energy measurements. Identical -per-job and collected artifact copies are allowed; conflicting copies fail. -Matching uses the ingest mapper's canonical identity, including AgentX `users` -precedence over `conc`. After benchmark writes, any required point omitted by a -purge or another filter fails the run; purged data is never restored to satisfy -the declaration. - -The manifest must name the source run and head. A successful earlier attempt of -that same run may supply the scope and retained points when failed jobs are -rerun; ingestion logs both declared and current attempts. Sweeps without this -optional manifest keep historical behavior. The separate PowerX publication -receipt compares ingested 8K/1K and AgentX measurements with the database and -public API after cache invalidation; it does not assert browser rendering. +Required ordinary sweeps upload a versioned +`required-power-sweep-manifest/sweep_manifest.json`. The [shared v2 fixture and +contract](./fixtures/powerx-manifest-v2/README.md) bind the complete required +matrix to source run/head/attempt, point identities, topology, exact measurement +windows, physical node/GPU roles and hashed evidence. Both repositories test the +same bytes. Unversioned required manifests fail closed; optional legacy bundles +retain their existing behavior. + +Artifact preparation validates required evidence before workflow migrations. +Required intent also travels in the dispatch payload, so losing both the manifest +and changelog marker cannot downgrade an ordinary required dispatch. Ingestion +repeats validation before workflow/config upserts, checks purges/backfills before +writing, and projects the resulting published curves from base-table state. It +models the actual latest-attempt, whole-curve and same-image append-only rules. +An unexplained loss of an existing recipe or concurrency point rejects ingestion. +Destructive replacement requires exact old-snapshot and lost-point identities in +the manifest; the ordinary producer supplies no such permission. + +The source run and head must match. A successful earlier attempt of that same run +may supply retained evidence when failed jobs are rerun; ingestion logs declared +and current attempts. Required energy must be finite and positive. Missing, +invalid and measured zero remain different values even though all fail this gate. +Disaggregated deployments require physical evidence for both roles. + +This is pure preflight, not atomic publication. Schema migrations occur after +artifact validation but before the curve check. Concurrent writers and failures +during the existing per-file importer remain a risk; staging plus an atomic, +serialized promotion is the follow-up described in the contract. The separate +PowerX receipt still compares source measurements against the DB and exact-run +API after ingestion. It does not prove latest-curve visibility or browser rendering. ### Append-Only Curve Extensions diff --git a/docs/fixtures/powerx-manifest-v2/README.md b/docs/fixtures/powerx-manifest-v2/README.md new file mode 100644 index 000000000..0075b4a13 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/README.md @@ -0,0 +1,101 @@ +# PowerX source manifest v2 + +This small, synthetic one-GPU AgentX bundle is copied byte-for-byte between +InferenceX and InferenceX-app. It tests the file contract. It is **not** a GPU +capture, hardware qualification or publication receipt. + +`artifacts/required-power-sweep-manifest/sweep_manifest.json` is the source +manifest. Evidence paths are relative to `artifacts/`, include their GitHub +artifact directory, and have SHA-256 hashes of the exact retained bytes. + +## Contract + +- `schema-version: 2` versions publication independently of measurement-row + `power_metric_schema_version: 2`. Unsupported or unversioned required bundles + fail closed; legacy optional bundles remain optional. +- `run-id`, `run-attempt`, and `head` name the tested source. A later rerun may + retain successful earlier-attempt evidence from the same run and head. The + consumer compares these fields with GitHub source metadata, including on reuse. +- `matrix` retains the ordinary planner declaration. Every required, + non-evaluation matrix point must occur exactly once in `points`. +- `config_key` is the matrix entry's `exp-name`. `identity` binds model, hardware, + framework, precision, recipe fingerprint, workload/sequence and concurrency. + AgentX uses `agentic_traces` with null `isl`/`osl`; fixed sequence uses + `single_turn`. Full recipe details remain in the matrix and hashed aggregate. +- `topology` declares aggregate versus P/D, single versus multi-node and physical + GPU counts. `devices` binds actual nodes and GPU UUIDs to aggregate, prefill or + decode roles with positive per-device energy. Both P/D roles are mandatory for + disaggregated deployments. A local GPU index alone is insufficient. +- `measurement_window` contains finite Unix-second boundaries, with end after + start. The evidence must describe the same window. +- `artifacts` contains confined relative paths, exact hashes and + `validation_state: "valid"`. Missing files, path escapes, invalid verdicts, + mismatched identities and invalid power reject required publication. Missing + and invalid values remain distinct from numeric zero. Required energy and + derived power must be strictly positive; measured zero fails that gate. + +## Publication intent + +Normal manifests emit `publication: {"mode":"incremental","replacement_scope":[]}`. +Incremental means **no authorized loss**; it does not change existing whole-curve +selection or automatically enable append-only. A complete refresh can pass by +retaining all stable point identities. Recipe/image fingerprint changes may need +a reviewed exact replacement. + +`mode: "replacement"` permits only explicit entries containing: + +- `curve_scope`: the canonical logical curve JSON identity; +- `previous_snapshot_workflow_run_id`: the exact database snapshot superseded; +- `removed_point_identities`: the exact set of lost stable identities. + +No wildcard, stale snapshot or broader permission is implied. The producer's +ordinary path emits no destructive authorization. Authoring authorization needs +explicitly reviewed replacement intent and current snapshot evidence. + +AgentX scope groups model/hardware/framework/precision/workload; AGG/P/D, +parallelism, offload and recipe variants are points in that curve. Fixed-sequence +scopes retain existing spec/disagg/offload boundaries. The consumer models latest +attempts, whole snapshots and same-image append-only inheritance before its first +publication write. Row counts or a stale materialized view cannot establish safety. + +## Remaining publication boundary + +Artifact validation runs before workflow migrations. Curve preflight runs before +workflow/config/benchmark upserts. It is a pure preflight, not an atomic commit: +concurrent writers and failures after the first write can still cause partial +publication. Follow-up work must stage the complete run, recheck the base-table +curve under a target-database publication lock, then atomically promote metadata +and benchmark rows. Sidecar preparation belongs outside that transaction. +Post-write DB/API checks remain necessary; exact-run equality alone does not prove +latest-curve visibility. + +## Read-only fixture check in InferenceX-app + +From the repository root: + +```sh +bun packages/db/src/preflight-power-publication.ts \ + docs/fixtures/powerx-manifest-v2/artifacts 123 1 \ + bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb +``` + +This opens no database connection, dispatches no work and writes no files. It +validates artifacts only; focused curve tests model previous published state. + +## 中文说明 + +此目录是一份最小的合成 AgentX 样例,两个仓库保留完全相同的文件字节, +用于测试 producer 与 consumer 的契约;它不是实机采集或生产发布证据。 + +manifest schema v2 记录来源 run、attempt、测试 SHA、完整 matrix、逐点身份、 +测量窗口、实际 node/GPU UUID/角色及产物哈希。必需功耗路径拒绝缺失、损坏、 +不兼容或能量非正的证据;P/D 必须同时覆盖 prefill 和 decode。历史可选功耗 +路径保留兼容行为。数值 0 不会被改写为“缺失”,但不能通过正能量门槛。 + +默认 incremental 只表示没有授权丢点,不会改变数据库整条曲线替换的语义。 +有意替换必须精确指定旧快照及全部移除点,不能使用通配或过期授权。AgentX +的 AGG、P/D、offload 和配方属于同一条逻辑曲线,校验时必须一起比较。 + +产物校验先于 migration;曲线校验先于发布数据写入。当前实现不是原子发布, +仍需后续 staging、发布锁和事务性 promote 来处理并发及中途写入失败。 +DB/API 写后校验仍保留,exactRun 校验不能单独证明 latest 曲线完整可见。 diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics.csv b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics.csv new file mode 100644 index 000000000..d39b2974a --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics.csv @@ -0,0 +1,3 @@ +timestamp, index, power.draw [W] +2023/11/14 22:13:20.000, 0, 500 +2023/11/14 22:13:22.000, 0, 500 diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics_identity.csv b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics_identity.csv new file mode 100644 index 000000000..139b5a261 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics_identity.csv @@ -0,0 +1,2 @@ +index, uuid, pci.bus_id, name, driver_version +0, GPU-golden, 00000000:01:00.0, NVIDIA H100, 000.00 diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_node.txt b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_node.txt new file mode 100644 index 000000000..fb271d435 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_node.txt @@ -0,0 +1 @@ +golden-node diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_validation.json b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_validation.json new file mode 100644 index 000000000..42d164cb2 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_validation.json @@ -0,0 +1,21 @@ +{ + "schema_version": 1, + "power_valid": true, + "reasons": [], + "benchmark_window": { + "start_time_unix": 1700000000, + "end_time_unix": 1700000002 + }, + "expected_gpu_count": 1, + "observed_gpu_count": 1, + "observed_gpu_ids": ["0"], + "per_gpu_energy_j": { + "0": 1000 + }, + "metrics": { + "avg_power_w": 500, + "avg_total_gpu_power_w": 500, + "total_gpu_energy_j": 1000, + "joules_per_output_token": 2 + } +} diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/bmk_agentic_golden/agg.json b/docs/fixtures/powerx-manifest-v2/artifacts/bmk_agentic_golden/agg.json new file mode 100644 index 000000000..98df7c14b --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/bmk_agentic_golden/agg.json @@ -0,0 +1,25 @@ +{ + "infmax_model_prefix": "qwen3.5", + "hw": "h100", + "framework": "sglang", + "precision": "fp8", + "recipe_fingerprint": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "scenario_type": "agentic-coding", + "conc": 1, + "users": 1, + "tp": 1, + "ep": 1, + "dp_attention": false, + "disagg": false, + "is_multinode": false, + "num_gpus": 1, + "image": "example/serving:golden", + "power_valid": 1, + "power_metric_schema_version": 2, + "avg_power_w": 500, + "avg_total_gpu_power_w": 500, + "total_gpu_energy_j": 1000, + "joules_per_output_token": 2, + "output_tput_per_gpu": 250, + "median_intvty": 50 +} diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/changelog-metadata/changelog_metadata.json b/docs/fixtures/powerx-manifest-v2/artifacts/changelog-metadata/changelog_metadata.json new file mode 100644 index 000000000..af53bb83d --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/changelog-metadata/changelog_metadata.json @@ -0,0 +1,4 @@ +{ + "require-power": true, + "entries": [] +} diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/required-power-sweep-manifest/sweep_manifest.json b/docs/fixtures/powerx-manifest-v2/artifacts/required-power-sweep-manifest/sweep_manifest.json new file mode 100644 index 000000000..f51df0546 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/required-power-sweep-manifest/sweep_manifest.json @@ -0,0 +1,98 @@ +{ + "schema-version": 2, + "head": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "run-id": 123, + "run-attempt": 1, + "full-sweep": false, + "matrix": { + "single_node": { + "agentic": [ + { + "exp-name": "golden", + "model-prefix": "qwen3.5", + "model": "Qwen/Qwen3.5", + "runner": "h100", + "framework": "sglang", + "precision": "fp8", + "recipe-fingerprint": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "conc": [1], + "require-power": true, + "scenario-type": "agentic-coding", + "disagg": false, + "num-gpus": 1, + "tp": 1, + "ep": 1, + "image": "example/serving:golden" + } + ] + }, + "multi_node": {} + }, + "publication": { + "mode": "incremental", + "replacement_scope": [] + }, + "points": [ + { + "config_key": "golden", + "identity": { + "model": "qwen3.5", + "hardware": "h100", + "framework": "sglang", + "precision": "fp8", + "recipe_fingerprint": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "benchmark_type": "agentic_traces", + "isl": null, + "osl": null, + "concurrency": 1 + }, + "topology": { + "disagg": false, + "is_multinode": false, + "num_gpus": 1, + "tp": 1, + "ep": 1, + "dp_attention": false + }, + "measurement_window": { + "start_time_unix": 1700000000, + "end_time_unix": 1700000002 + }, + "devices": [ + { + "node": "golden-node", + "gpu_uuid": "GPU-golden", + "role": "aggregate", + "energy_j": 1000 + } + ], + "artifacts": [ + { + "path": "agentic_golden/gpu_metrics.csv", + "sha256": "f42dee832f8d837142be9d4dc5fbbf8462c02846f06e66edfce659b8b20943db", + "validation_state": "valid" + }, + { + "path": "agentic_golden/gpu_metrics_identity.csv", + "sha256": "a4364a0d4050078b9cba24c7e695fbed658fabb1d48e61a3da8f61b8ec873627", + "validation_state": "valid" + }, + { + "path": "agentic_golden/power_node.txt", + "sha256": "114aa18bee3688c0e6040020ad0230472645541372b88a54253116fd9c2148e6", + "validation_state": "valid" + }, + { + "path": "agentic_golden/power_validation.json", + "sha256": "c00687e7239fd172aa15f60c8d967294a0fa623984d0eaeeb5222424b6335d94", + "validation_state": "valid" + }, + { + "path": "bmk_agentic_golden/agg.json", + "sha256": "56e3377ae12402eec61e25e2f9a0480ab228048a29d32c35cb8d438d50ea2faa", + "validation_state": "valid" + } + ] + } + ] +} diff --git a/packages/db/src/etl/power-publication.test.ts b/packages/db/src/etl/power-publication.test.ts index f223a07a4..b31ce0f98 100644 --- a/packages/db/src/etl/power-publication.test.ts +++ b/packages/db/src/etl/power-publication.test.ts @@ -37,7 +37,7 @@ function expected(overrides = {}) { 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/123/attempts/2', { path: 'bmk_qwen3.5/results.json', sha256: 'abc' }, ); - if (!point) throw new Error('Fixture must be 8K/1K'); + if (!point) throw new Error('Fixture must belong to a supported PowerX workload'); return point; } function actual(point = expected()): PublishedPowerRow { @@ -51,6 +51,12 @@ function actual(point = expected()): PublishedPowerRow { } describe('PowerX publication', () => { + it('keeps required 1K/1K measurements in the publication receipt', () => { + const point = expected({ isl: 1024, joules_per_output_token: 2.5 }); + expect(point.identity).toMatchObject({ benchmark_type: 'single_turn', isl: 1024, osl: 1024 }); + expect(verifyPowerPublication([point], [actual(point)], 'database')).toEqual([]); + expect(verifyPowerPublication([point], [], 'public API')[0]).toContain('found 0'); + }); it('verifies AgentX source identity, nullable sequences, energy and audit through DB/API', () => { const point = expected({ scenario_type: 'agentic-coding', diff --git a/packages/db/src/etl/power-publication.ts b/packages/db/src/etl/power-publication.ts index d2ed10c14..5009468be 100644 --- a/packages/db/src/etl/power-publication.ts +++ b/packages/db/src/etl/power-publication.ts @@ -52,20 +52,22 @@ export interface PublishedPowerRow extends Record { metrics: Record; } +export function stablePowerPointIdentity(row: Record): string { + return JSON.stringify( + IDENTITY_FIELDS.filter((key) => key !== 'image' && key !== 'run_url').map( + (key) => row[key] ?? null, + ), + ); +} + export function publicationIdentity(row: Record): string { return JSON.stringify(IDENTITY_FIELDS.map((key) => row[key] ?? null)); } -export function powerPublicationPoint( +export function benchmarkPublicationIdentity( row: BenchmarkParams, - runUrl: string, - artifact: PowerPublicationPoint['artifact'], -): PowerPublicationPoint | null { - if ( - row.benchmarkType !== 'agentic_traces' && - (row.benchmarkType !== 'single_turn' || row.isl !== 8192 || row.osl !== 1024) - ) - return null; + runUrl = '', +): Record { const identity: Record = Object.fromEntries( Object.entries(CONFIG_FIELDS).map(([source, target]) => [ target, @@ -82,6 +84,22 @@ export function powerPublicationPoint( image: row.image, run_url: runUrl, }); + return identity; +} + +export function powerPublicationPoint( + row: BenchmarkParams, + runUrl: string, + artifact: PowerPublicationPoint['artifact'], +): PowerPublicationPoint | null { + if ( + row.benchmarkType !== 'agentic_traces' && + (row.benchmarkType !== 'single_turn' || + (row.isl !== 1024 && row.isl !== 8192) || + row.osl !== 1024) + ) + return null; + const identity = benchmarkPublicationIdentity(row, runUrl); return { identity, metrics: Object.fromEntries( diff --git a/packages/db/src/etl/required-power-curve-db.test.ts b/packages/db/src/etl/required-power-curve-db.test.ts new file mode 100644 index 000000000..a5ec3da12 --- /dev/null +++ b/packages/db/src/etl/required-power-curve-db.test.ts @@ -0,0 +1,133 @@ +import { PGlite } from '@electric-sql/pglite'; +import fs from 'node:fs'; +import path from 'node:path'; +import os from 'node:os'; +import { beforeAll, beforeEach, afterAll, describe, it, expect } from 'vitest'; +import type { DbClient } from '../connection'; +import { preflightRequiredPowerCurves } from './required-power-curve'; +import { getLatestBenchmarks } from '../queries/benchmarks'; +let db: PGlite; +const sql: DbClient = async (strings, ...values) => { + const query = strings.reduce((text, part, index) => text + (index ? `$${index}` : '') + part, ''); + const result = await db.query>(query, values); + return result.rows; +}; +const golden = path.resolve( + import.meta.dirname, + '../../../../docs/fixtures/powerx-manifest-v2/artifacts', +); +const source = { runId: 123, runAttempt: 1, headSha: 'b'.repeat(40) }; +const options = { date: '2026-09-16', runStartedAt: '2026-09-16T00:00:00Z', appendOnly: false }; +beforeAll(async () => { + db = await PGlite.create(); + const dir = new URL('../../migrations/', import.meta.url); + for (const file of fs + .readdirSync(dir) + .filter((name) => name.endsWith('.sql')) + .sort()) + await db.exec(fs.readFileSync(new URL(file, dir), 'utf8')); +}, 20000); +afterAll(async () => { + await db?.close(); +}); +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + prefill_tp, prefill_ep, decode_tp, decode_ep, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'qwen3.5', 'h100', 'sglang', 'fp8', 'none', false, 1,1,1,1,1,1); + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, date, created_at, run_started_at) + VALUES (1, 111, 1, 'previous', '2026-09-15', '2026-09-15T00:00:00Z', '2026-09-15T00:00:00Z');`); +}); +async function addPoint(conc: number, image = 'example/serving:golden') { + await sql`INSERT INTO benchmark_results (config_id, workflow_run_id, date, benchmark_type, isl, osl, conc, offload_mode, recipe_fingerprint, image, metrics) + VALUES (1,1,'2026-09-15','agentic_traces',NULL,NULL,${conc},'off',${'a'.repeat(64)},${image},'{}')`; +} +describe('read-only required-power DB preflight', () => { + it('rejects partial refresh from base tables before any write, even with a stale latest view', async () => { + await addPoint(1); + await addPoint(64); + const before = await sql`SELECT count(*)::int AS n FROM benchmark_results`; + await expect(preflightRequiredPowerCurves(sql, golden, source, options)).rejects.toThrow( + 'shrink', + ); + expect(await sql`SELECT count(*)::int AS n FROM benchmark_results`).toEqual(before); + expect(await sql`SELECT count(*)::int AS n FROM workflow_runs`).toEqual([{ n: 1 }]); + const published = await getLatestBenchmarks(sql, 'qwen3.5', '9999-12-31'); + expect(published.map((row) => row.conc)).toEqual([1, 64]); + }); + it('accepts complete incoming point coverage', async () => { + await addPoint(1); + await expect( + preflightRequiredPowerCurves(sql, golden, source, options), + ).resolves.toBeUndefined(); + expect(await sql`SELECT count(*)::int AS n FROM workflow_runs`).toEqual([{ n: 1 }]); + }); + it('inherits same-image append-only state but rejects a changed image', async () => { + await addPoint(1); + await addPoint(64); + await expect( + preflightRequiredPowerCurves(sql, golden, source, { ...options, appendOnly: true }), + ).resolves.toBeUndefined(); + await sql`UPDATE benchmark_results SET image='old-image'`; + await expect( + preflightRequiredPowerCurves(sql, golden, source, { ...options, appendOnly: true }), + ).rejects.toThrow('shrink'); + }); + it('detects retry removal from a scope omitted entirely by incoming artifacts', async () => { + await sql`UPDATE workflow_runs SET github_run_id=123`; + await sql`UPDATE configs SET hardware='h200'`; + await addPoint(64); + await expect( + preflightRequiredPowerCurves(sql, golden, { ...source, runAttempt: 2 }, options), + ).rejects.toThrow('shrink'); + expect(await sql`SELECT count(*)::int AS n FROM workflow_runs`).toEqual([{ n: 1 }]); + }); + it('protects optional fixed workloads outside the power receipt whitelist', async () => { + await sql`INSERT INTO benchmark_results (config_id,workflow_run_id,date,benchmark_type,isl,osl,conc,offload_mode,recipe_fingerprint,image,metrics) + VALUES (1,1,'2026-09-15','single_turn',4096,1024,1,'off',${'c'.repeat(64)},'example/serving:golden','{}'), + (1,1,'2026-09-15','single_turn',4096,1024,8,'off',${'c'.repeat(64)},'example/serving:golden','{}')`; + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'powerx-optional-')); + try { + fs.cpSync(golden, dir, { recursive: true }); + fs.mkdirSync(path.join(dir, 'results_optional')); + const row = JSON.parse( + fs.readFileSync(path.join(golden, 'bmk_agentic_golden/agg.json'), 'utf8'), + ); + delete row.scenario_type; + delete row.users; + Object.assign(row, { isl: 4096, osl: 1024, recipe_fingerprint: 'c'.repeat(64) }); + fs.writeFileSync(path.join(dir, 'results_optional/extra.json'), JSON.stringify(row)); + await expect(preflightRequiredPowerCurves(sql, dir, source, options)).rejects.toThrow( + 'shrink', + ); + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); + it('models mixed dates on same-attempt upserts instead of overlooking a newer snapshot', async () => { + await sql`UPDATE workflow_runs SET github_run_id=123`; + await addPoint(1); + await sql`UPDATE benchmark_results SET date='2026-09-10'`; + await sql`INSERT INTO workflow_runs (id,github_run_id,run_attempt,name,date,created_at,run_started_at) + VALUES (2,222,1,'competing','2026-09-12','2026-09-12T00:00:00Z','2026-09-12T00:00:00Z')`; + await sql`INSERT INTO benchmark_results (config_id,workflow_run_id,date,benchmark_type,isl,osl,conc,offload_mode,recipe_fingerprint,image,metrics) + VALUES (1,2,'2026-09-12','agentic_traces',NULL,NULL,8,'off',${'a'.repeat(64)},'example/serving:golden','{}')`; + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'powerx-dates-')); + try { + fs.cpSync(golden, dir, { recursive: true }); + fs.mkdirSync(path.join(dir, 'results_optional')); + const row = JSON.parse( + fs.readFileSync(path.join(golden, 'bmk_agentic_golden/agg.json'), 'utf8'), + ); + Object.assign(row, { conc: 4, users: 4 }); + fs.writeFileSync(path.join(dir, 'results_optional/extra.json'), JSON.stringify(row)); + await expect(preflightRequiredPowerCurves(sql, dir, source, options)).rejects.toThrow( + 'shrink', + ); + const published = await getLatestBenchmarks(sql, 'qwen3.5', '9999-12-31'); + expect(published.map((point) => point.conc)).toEqual([8]); + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); +}); diff --git a/packages/db/src/etl/required-power-curve.test.ts b/packages/db/src/etl/required-power-curve.test.ts new file mode 100644 index 000000000..b94378c09 --- /dev/null +++ b/packages/db/src/etl/required-power-curve.test.ts @@ -0,0 +1,123 @@ +import { describe, expect, it } from 'vitest'; +import { + assertCurvePreserved, + publishedCurve, + type CurvePoint, + type CurvePublication, +} from './required-power-curve'; +import { powerPublicationPoint, stablePowerPointIdentity } from './power-publication'; +import { verifyRequiredPowerArtifacts } from './required-power-publication'; +import path from 'node:path'; +const golden = path.resolve( + import.meta.dirname, + '../../../../docs/fixtures/powerx-manifest-v2/artifacts', +); +const benchmark = verifyRequiredPowerArtifacts(golden, { + runId: 123, + runAttempt: 1, + headSha: 'b'.repeat(40), +})[0]; +const identity = powerPublicationPoint(benchmark, '', { path: '', sha256: '' })!.identity; +const policy: CurvePublication = { mode: 'incremental', replacement_scope: [] }; +function point(run: number, conc: number, extra: Partial = {}): CurvePoint { + return { + identity: { ...identity, conc }, + image: 'same-image', + workflowRunId: run, + githubRunId: run, + runAttempt: 1, + date: `2026-09-${run.toString().padStart(2, '0')}`, + runStartedAt: null, + appendOnly: false, + ...extra, + }; +} +describe('pure projected publication curve', () => { + it('rejects the old partial-sweep regression without altering input rows', () => { + const old = [ + point(1, 1), + point(1, 16), + point(1, 64, { + identity: { ...identity, conc: 64, disagg: true, recipe_fingerprint: 'split-recipe' }, + }), + ]; + const proposed = [...old, point(2, 1)]; + expect(() => assertCurvePreserved(old, proposed, policy)).toThrow('shrink'); + expect([...publishedCurve(old).values()][0]).toHaveLength(3); + }); + it('accepts a complete refresh and unrelated curve additions', () => { + const old = [point(1, 1), point(1, 16)]; + expect(() => + assertCurvePreserved(old, [...old, point(2, 1), point(2, 16)], policy), + ).not.toThrow(); + expect(() => + assertCurvePreserved( + old, + [...old, point(2, 1, { identity: { ...identity, hardware: 'h200' } })], + policy, + ), + ).not.toThrow(); + }); + it.each([null, 'new-image'])( + 'does not inherit append-only history with incompatible image %s', + (image) => { + const old = [point(1, 1), point(1, 16)]; + expect(() => + assertCurvePreserved(old, [...old, point(2, 32, { appendOnly: true, image })], policy), + ).toThrow('shrink'); + }, + ); + it('inherits only an uninterrupted same-image append-only chain', () => { + const old = [point(1, 1), point(1, 16), point(2, 32, { appendOnly: true })]; + expect([...publishedCurve(old).values()][0].map((row) => row.identity.conc)).toEqual([ + 32, 1, 16, + ]); + expect(() => + assertCurvePreserved(old, [...old, point(3, 64, { appendOnly: true })], policy), + ).not.toThrow(); + expect(() => assertCurvePreserved(old, [...old, point(3, 64)], policy)).toThrow('shrink'); + }); + it('permits only the exact removed identities of the observed snapshot', () => { + const old = [point(1, 1), point(1, 16)]; + const proposed = [...old, point(2, 1)]; + const replacement: CurvePublication = { + mode: 'replacement', + replacement_scope: [ + { + curve_scope: [...publishedCurve(old).keys()][0], + previous_snapshot_workflow_run_id: 1, + removed_point_identities: [stablePowerPointIdentity(old[1].identity)], + }, + ], + }; + expect(() => assertCurvePreserved(old, proposed, replacement)).not.toThrow(); + for (const altered of [ + { ...replacement.replacement_scope[0], previous_snapshot_workflow_run_id: 99 }, + { ...replacement.replacement_scope[0], removed_point_identities: ['*'] }, + { ...replacement.replacement_scope[0], curve_scope: '*' }, + ]) + expect(() => + assertCurvePreserved(old, proposed, { mode: 'replacement', replacement_scope: [altered] }), + ).toThrow('shrink'); + }); + it('treats topology and offload as point identity inside one AgentX curve', () => { + const old = [point(1, 1)]; + for (const altered of [ + { prefill_tp: 2 }, + { offload_mode: 'on' }, + { recipe_fingerprint: 'other-recipe' }, + ]) { + expect(() => + assertCurvePreserved( + old, + [...old, point(2, 1, { identity: { ...identity, ...altered } })], + policy, + ), + ).toThrow('shrink'); + } + }); + it('does not let an older run replace newer curve state', () => { + const old = [point(2, 1), point(2, 16)]; + expect(() => assertCurvePreserved(old, [...old, point(1, 1)], policy)).not.toThrow(); + }); +}); diff --git a/packages/db/src/etl/required-power-curve.ts b/packages/db/src/etl/required-power-curve.ts new file mode 100644 index 000000000..1b2cdfbfa --- /dev/null +++ b/packages/db/src/etl/required-power-curve.ts @@ -0,0 +1,287 @@ +import fs from 'node:fs'; +import path from 'node:path'; +import { + benchmarkCurveScope, + type BenchmarkCurveInput, +} from '@semianalysisai/inferencex-constants'; +import type { DbClient } from '../connection'; +import type { Sql } from './db-utils'; +import { REQUIRED_POWER_MANIFEST } from '../lib/ci-artifact-preparation'; +import { mapBenchmarkRow, type BenchmarkParams } from './benchmark-mapper'; +import { configCacheKey, type ConfigParams } from './config-cache'; +import { benchmarkPublicationIdentity, stablePowerPointIdentity } from './power-publication'; +import { + assertRequiredPowerPointsRetained, + verifyRequiredPowerArtifacts, + type RequiredPowerSource, +} from './required-power-publication'; +import { + applyBenchmarkPointBackfill, + isBenchmarkPointPurged, + recordBackfilledPointIdentity, +} from './run-overrides'; +import { createSkipTracker } from './skip-tracker'; + +export interface CurvePoint { + identity: Record; + image: string | null; + workflowRunId: number; + githubRunId: number; + runAttempt: number; + date: string; + runStartedAt: string | null; + appendOnly: boolean; +} +export interface CurveReplacement { + curve_scope: string; + previous_snapshot_workflow_run_id: number; + removed_point_identities: string[]; +} +export interface CurvePublication { + mode: 'incremental' | 'replacement'; + replacement_scope: CurveReplacement[]; +} + +function scope(point: CurvePoint): string { + return benchmarkCurveScope(point.identity as unknown as BenchmarkCurveInput); +} + +/** Mirrors migration 014: latest attempt, scope-level snapshots and same-image append chains. */ +export function publishedCurve(points: readonly CurvePoint[]): Map { + const latestAttempts = new Map(); + for (const point of points) + latestAttempts.set( + point.githubRunId, + Math.max(latestAttempts.get(point.githubRunId) ?? 0, point.runAttempt), + ); + const scopes = new Map>(); + for (const point of points) { + if (point.runAttempt !== latestAttempts.get(point.githubRunId)) continue; + const key = scope(point); + const runs = scopes.get(key) ?? new Map(); + const runDate = JSON.stringify([point.workflowRunId, point.date]); + runs.set(runDate, [...(runs.get(runDate) ?? []), point]); + scopes.set(key, runs); + } + const result = new Map(); + for (const [key, runs] of scopes) { + const ranked = [...runs.values()].sort( + (a, b) => + b[0].date.localeCompare(a[0].date) || + (b[0].runStartedAt ? Date.parse(b[0].runStartedAt) : -Infinity) - + (a[0].runStartedAt ? Date.parse(a[0].runStartedAt) : -Infinity) || + b[0].workflowRunId - a[0].workflowRunId, + ); + const rootImage = ranked[0][0].image; + const uniformImage = (rows: CurvePoint[]) => + rootImage !== null && rows.every((row) => row.image === rootImage); + const selected = new Map(); + for (let index = 0; index < ranked.length; index++) { + const current = ranked[index]; + // SQL groups run/date for ranking, then joins every point in that run/scope. + for (const point of [...runs.values()] + .flat() + .filter((candidate) => candidate.workflowRunId === current[0].workflowRunId)) { + const id = stablePowerPointIdentity(point.identity); + if (!selected.has(id)) selected.set(id, point); + } + if ( + !current[0].appendOnly || + !uniformImage(current) || + !ranked[index + 1] || + !uniformImage(ranked[index + 1]) + ) + break; + } + result.set(key, [...selected.values()]); + } + return result; +} + +function canonical(entries: CurveReplacement[]): string { + return JSON.stringify( + entries + .map((entry) => { + if ( + typeof entry.curve_scope !== 'string' || + !Number.isSafeInteger(entry.previous_snapshot_workflow_run_id) || + !Array.isArray(entry.removed_point_identities) || + entry.removed_point_identities.some((id) => typeof id !== 'string') + ) + throw new Error('Required power: invalid exact replacement scope'); + return { ...entry, removed_point_identities: [...entry.removed_point_identities].sort() }; + }) + .sort((a, b) => a.curve_scope.localeCompare(b.curve_scope)), + ); +} + +/** Exact lost identities bind any destructive authorization to one observed snapshot. */ +export function assertCurvePreserved( + existing: readonly CurvePoint[], + proposed: readonly CurvePoint[], + publication: CurvePublication, +): void { + if ( + !publication || + !['incremental', 'replacement'].includes(publication.mode) || + !Array.isArray(publication.replacement_scope) + ) + throw new Error('Required power: invalid curve publication policy'); + const before = publishedCurve(existing); + const after = publishedCurve(proposed); + const losses: CurveReplacement[] = []; + for (const [key, oldPoints] of before) { + const next = new Set( + (after.get(key) ?? []).map((point) => stablePowerPointIdentity(point.identity)), + ); + const removed = oldPoints + .map((point) => stablePowerPointIdentity(point.identity)) + .filter((identity) => !next.has(identity)) + .sort(); + if (removed.length > 0) + losses.push({ + curve_scope: key, + previous_snapshot_workflow_run_id: oldPoints[0].workflowRunId, + removed_point_identities: removed, + }); + } + if (publication.mode === 'incremental' && publication.replacement_scope.length > 0) + throw new Error('Required power: incremental publication cannot authorize replacement'); + if ( + (losses.length > 0 && publication.mode !== 'replacement') || + canonical(losses) !== canonical(publication.replacement_scope) + ) + throw new Error( + `Required power: publication would shrink the existing curve; exact replacement_scope required: ${JSON.stringify(losses)}`, + ); +} + +/** Read-only preflight; no config/workflow upsert, migration, or materialized-view refresh. */ +export async function preflightRequiredPowerCurves( + sql: DbClient | Sql, + root: string, + source: RequiredPowerSource, + options: { date: string; runStartedAt: string | null; appendOnly: boolean }, +): Promise { + const required = verifyRequiredPowerArtifacts(root, source); + if (required.length === 0) return; + const manifest = JSON.parse( + fs.readFileSync(path.join(root, REQUIRED_POWER_MANIFEST, 'sweep_manifest.json'), 'utf8'), + ); + const configs = await sql`SELECT * FROM configs`; + const configIds = new Map( + configs.map((row) => [ + configCacheKey({ + hardware: row.hardware, + framework: row.framework, + model: row.model, + precision: row.precision, + specMethod: row.spec_method, + disagg: row.disagg, + isMultinode: row.is_multinode, + prefillTp: row.prefill_tp, + prefillEp: row.prefill_ep, + prefillDpAttn: row.prefill_dp_attention, + prefillNumWorkers: row.prefill_num_workers, + decodeTp: row.decode_tp, + decodeEp: row.decode_ep, + decodeDpAttn: row.decode_dp_attention, + decodeNumWorkers: row.decode_num_workers, + numPrefillGpu: row.num_prefill_gpu, + numDecodeGpu: row.num_decode_gpu, + } as ConfigParams), + Number(row.id), + ]), + ); + const incoming = new Map(); + const backfilled = new Map(); + for (const name of fs.readdirSync(root)) { + if ( + (!name.startsWith('bmk_') && !name.startsWith('results_')) || + !fs.statSync(path.join(root, name)).isDirectory() + ) + continue; + for (const file of fs + .readdirSync(path.join(root, name)) + .filter((candidateName) => candidateName.endsWith('.json'))) { + const data = JSON.parse(fs.readFileSync(path.join(root, name, file), 'utf8')); + for (const raw of Array.isArray(data) ? data : [data]) { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) continue; + const mapped = mapBenchmarkRow(raw, createSkipTracker(), undefined, source.runId); + if (!mapped) continue; + // A config not present yet cannot match a config-id-scoped purge/backfill. + const point = { ...mapped, configId: configIds.get(configCacheKey(mapped.config)) ?? -1 }; + if (isBenchmarkPointPurged(source.runId, source.runAttempt, point)) continue; + const applied = applyBenchmarkPointBackfill(source.runId, source.runAttempt, point); + recordBackfilledPointIdentity(backfilled, applied.sourceIdentity, applied.desiredIdentity); + incoming.set( + stablePowerPointIdentity(benchmarkPublicationIdentity(applied.point)), + applied.point, + ); + } + } + } + assertRequiredPowerPointsRetained(required, [...incoming.values()]); + const models = [...new Set([...incoming.values()].map((point) => point.config.model))]; + const rows = await sql` + SELECT c.*, br.benchmark_type, br.isl, br.osl, br.conc, br.offload_mode, + br.recipe_fingerprint, br.image, br.date::text AS point_date, + wr.id AS workflow_run_id, wr.github_run_id, wr.run_attempt, + wr.run_started_at::text, wr.append_only + FROM benchmark_results br + JOIN configs c ON c.id = br.config_id + JOIN latest_workflow_runs wr ON wr.id = br.workflow_run_id + WHERE br.error IS NULL AND (c.model = ANY(${models}) OR wr.github_run_id = ${source.runId})`; + const stored = rows.map((row): CurvePoint => ({ + identity: row, + image: row.image as string | null, + workflowRunId: Number(row.workflow_run_id), + githubRunId: Number(row.github_run_id), + runAttempt: Number(row.run_attempt), + date: String(row.point_date), + runStartedAt: row.run_started_at as string | null, + appendOnly: Boolean(row.append_only), + })); + const attempts = + await sql`SELECT id, run_attempt FROM workflow_runs WHERE github_run_id = ${source.runId}`; + const sameAttempt = attempts.find((row) => Number(row.run_attempt) === source.runAttempt); + // New serial IDs sort after stored IDs. A concurrent ingest is outside this pure preflight boundary. + const workflowRunId = sameAttempt ? Number(sameAttempt.id) : Number.MAX_SAFE_INTEGER; + const newPoints = [...incoming.values()].map((point): CurvePoint => ({ + identity: benchmarkPublicationIdentity(point), + image: point.image, + workflowRunId, + githubRunId: source.runId, + runAttempt: source.runAttempt, + date: + stored.find( + (existing) => + existing.workflowRunId === workflowRunId && + stablePowerPointIdentity(existing.identity) === + stablePowerPointIdentity(benchmarkPublicationIdentity(point)), + )?.date ?? options.date, + runStartedAt: options.runStartedAt, + appendOnly: options.appendOnly, + })); + const touched = new Set( + [...newPoints, ...stored.filter((point) => point.githubRunId === source.runId)].map(scope), + ); + const existing = stored.filter((point) => touched.has(scope(point))); + const maxAttempt = Math.max(source.runAttempt, ...attempts.map((row) => Number(row.run_attempt))); + const changedIds = new Set(newPoints.map((point) => stablePowerPointIdentity(point.identity))); + const proposed = existing + .filter( + (point) => + point.githubRunId !== source.runId || + (point.runAttempt === maxAttempt && + (source.runAttempt !== maxAttempt || + !changedIds.has(stablePowerPointIdentity(point.identity)))), + ) + .map((point) => + point.githubRunId === source.runId && point.runAttempt === source.runAttempt + ? { ...point, runStartedAt: options.runStartedAt, appendOnly: options.appendOnly } + : point, + ); + if (source.runAttempt === maxAttempt) proposed.push(...newPoints); + assertCurvePreserved(existing, proposed, manifest.publication); +} diff --git a/packages/db/src/etl/required-power-publication.test.ts b/packages/db/src/etl/required-power-publication.test.ts index b51b8e526..7ad044559 100644 --- a/packages/db/src/etl/required-power-publication.test.ts +++ b/packages/db/src/etl/required-power-publication.test.ts @@ -1,314 +1,322 @@ +import { createHash } from 'node:crypto'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { afterEach, describe, expect, it } from 'vitest'; -import agenticMatrix from './__fixtures__/required-power-agentic-matrix.json'; import { assertRequiredPowerPointsRetained, verifyRequiredPowerArtifacts, - verifyRequiredPowerPublication, } from './required-power-publication'; -import { - isBenchmarkPointPurged, - PURGED_BENCHMARK_POINTS, - type PurgedBenchmarkPoint, -} from './run-overrides'; -// Ordinary planner output for Qwen3.5/H100 8K1K; fingerprints include the full recipe. -const fingerprint = 'fa0645b4dec181eafda7472c891e90d9a5bf7bad7ee764aa0fdcc41ff841adf1'; -const source = { runId: 123, runAttempt: 2, headSha: 'abc123' }; -const required = { - 'recipe-fingerprint': fingerprint, - conc: 1, - isl: 8192, - osl: 1024, - 'require-power': true, -}; -const row = { - infmax_model_prefix: 'qwen3.5', - hw: 'h100', - framework: 'sglang', - precision: 'fp8', - recipe_fingerprint: fingerprint, - conc: 1, - isl: 8192, - osl: 1024, - tp: 8, - ep: 1, - power_valid: 1, - power_metric_schema_version: 2, - avg_power_w: 500, - avg_total_gpu_power_w: 4000, - total_gpu_energy_j: 8000, - joules_per_output_token: 2, -}; -const manifest = (entries = [required]) => ({ - head: source.headSha, - 'run-id': source.runId, - 'run-attempt': source.runAttempt, - // full-sweep is a Klaud policy field: fail-fast runs may legitimately set it false. - 'full-sweep': false, - matrix: { - single_node: { '8k1k': entries }, - multi_node: {}, - evals: [{ ...required, 'eval-only': true }], - }, -}); -const artifacts = (rows: unknown[] = [row]) => [{ path: 'bmk_qwen/agg.json', rows }]; +const golden = path.resolve( + import.meta.dirname, + '../../../../docs/fixtures/powerx-manifest-v2/artifacts', +); +const source = { runId: 123, runAttempt: 1, headSha: 'b'.repeat(40) }; const dirs: string[] = []; +function fixture() { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'powerx-contract-')); + dirs.push(dir); + fs.cpSync(golden, dir, { recursive: true }); + return dir; +} +function json(dir: string, file: string): any { + return JSON.parse(fs.readFileSync(path.join(dir, file), 'utf8')); +} +function write(dir: string, file: string, value: unknown) { + fs.writeFileSync(path.join(dir, file), JSON.stringify(value)); +} +const manifestPath = 'required-power-sweep-manifest/sweep_manifest.json'; +const benchmarkPath = 'bmk_agentic_golden/agg.json'; +const auditPath = 'agentic_golden/power_validation.json'; +function changeManifest(dir: string, edit: (manifest: any) => void) { + const manifest = json(dir, manifestPath); + edit(manifest); + write(dir, manifestPath, manifest); +} +function changeArtifact(dir: string, file: string, edit: (value: any) => void) { + const value = json(dir, file); + edit(value); + write(dir, file, value); + changeManifest(dir, (manifest) => { + for (const point of manifest.points) + for (const artifact of point.artifacts) + if (artifact.path === file) + artifact.sha256 = createHash('sha256') + .update(fs.readFileSync(path.join(dir, file))) + .digest('hex'); + }); +} +function multinodeFixture() { + const dir = fixture(); + const extras: Record = { + 'power_audit_golden/power/samples.csv': + 'timestamp_unix,hostname,gpu_uuid,power_w\n1700000000,prefill,GPU-p,200\n', + 'power_audit_golden/power/manifest.json': { status: 'complete', publication_valid: true }, + 'power_audit_golden/power/windows/point.json': { + benchmark_start_time_unix: 1700000000, + benchmark_end_time_unix: 1700000002, + }, + }; + for (const [file, value] of Object.entries(extras)) { + fs.mkdirSync(path.dirname(path.join(dir, file)), { recursive: true }); + fs.writeFileSync( + path.join(dir, file), + typeof value === 'string' ? value : JSON.stringify(value), + ); + } + changeArtifact(dir, benchmarkPath, (row) => + Object.assign(row, { + disagg: true, + is_multinode: true, + num_gpus: 2, + num_prefill_gpu: 1, + num_decode_gpu: 1, + prefill_gpu_energy_j: 400, + decode_gpu_energy_j: 600, + prefill_joules_per_input_token: 1, + decode_joules_per_output_token: 1, + }), + ); + changeArtifact(dir, auditPath, (audit) => + Object.assign(audit, { + expected_gpu_count: 2, + observed_gpu_count: 2, + per_gpu_energy_j: { 'prefill/GPU-p': 400, 'decode/GPU-d': 600 }, + per_gpu_role: { 'prefill/GPU-p': 'prefill', 'decode/GPU-d': 'decode' }, + }), + ); + changeManifest(dir, (manifest) => { + const entry = manifest.matrix.single_node.agentic[0]; + Object.assign(entry, { disagg: true, 'num-gpus': 2, 'node-count': 2 }); + manifest.matrix = { single_node: {}, multi_node: { agentic: [entry] } }; + const point = manifest.points[0]; + Object.assign(point.topology, { + disagg: true, + is_multinode: true, + num_gpus: 2, + num_prefill_gpu: 1, + num_decode_gpu: 1, + }); + point.devices = [ + { node: 'prefill', gpu_uuid: 'GPU-p', role: 'prefill', energy_j: 400 }, + { node: 'decode', gpu_uuid: 'GPU-d', role: 'decode', energy_j: 600 }, + ]; + for (const file of Object.keys(extras)) + point.artifacts.push({ + path: file, + sha256: createHash('sha256') + .update(fs.readFileSync(path.join(dir, file))) + .digest('hex'), + validation_state: 'valid', + }); + }); + return dir; +} afterEach(() => { for (const dir of dirs.splice(0)) fs.rmSync(dir, { recursive: true, force: true }); }); -describe('required power publication preflight', () => { - it('matches the canonical AgentX users value instead of a conflicting raw conc', () => { - const scope = { - ...manifest(), - matrix: { single_node: { agentic: [required] }, multi_node: {} }, - }; - const conflicting = { ...row, scenario_type: 'agentic-coding', conc: 1, users: 2 }; - expect(() => verifyRequiredPowerPublication(scope, artifacts([conflicting]), source)).toThrow( - 'missing benchmark point', +describe('required power publication contract', () => { + it('accepts the shared producer golden fixture and an earlier declared attempt of the same head', () => { + expect(verifyRequiredPowerArtifacts(golden, source, true)).toMatchObject([ + { conc: 1, benchmarkType: 'agentic_traces', metrics: { total_gpu_energy_j: 1000 } }, + ]); + expect(verifyRequiredPowerArtifacts(golden, { ...source, runAttempt: 2 }, true)).toHaveLength( + 1, ); - expect( - verifyRequiredPowerPublication( - scope, - artifacts([{ ...conflicting, conc: 2, users: 1 }]), - source, - ), - ).toHaveLength(1); }); - - it('accepts exact ordinary source points, including fail-fast manifests and identical aggregate copies', () => { - expect( - verifyRequiredPowerPublication( - manifest(), - [...artifacts(), { path: 'results_bmk/agg_bmk.json', rows: [{ ...row }] }], - source, - ), - ).toHaveLength(1); + it('fails missing required manifests even when changelog metadata is also missing', () => { + const dir = fixture(); + fs.rmSync(path.join(dir, 'required-power-sweep-manifest'), { recursive: true }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('sweep manifest missing'); + fs.rmSync(path.join(dir, 'changelog-metadata'), { recursive: true }); + expect(() => verifyRequiredPowerArtifacts(dir, source, true)).toThrow('sweep manifest missing'); + expect(verifyRequiredPowerArtifacts(dir, source)).toEqual([]); }); - - it('fails required coverage after a point purge even when an optional point is retained', () => { - const requiredPoints = verifyRequiredPowerPublication(manifest(), artifacts(), source); - const candidates = [requiredPoints[0], { ...requiredPoints[0], conc: 2 }]; - const purged: PurgedBenchmarkPoint = { - githubRunId: source.runId, - runAttempt: source.runAttempt, - configId: 999999, - benchmarkType: 'single_turn', - isl: 8192, - osl: 1024, - conc: 1, - offloadMode: 'off', - recipeFingerprint: fingerprint, - }; - const registry = PURGED_BENCHMARK_POINTS as PurgedBenchmarkPoint[]; - registry.push(purged); - try { - const retained = candidates.filter( - (point) => - !isBenchmarkPointPurged(source.runId, source.runAttempt, { - ...point, - configId: purged.configId, - }), - ); - expect(retained.map((point) => point.conc)).toEqual([2]); - expect(() => assertRequiredPowerPointsRetained(requiredPoints, retained)).toThrow( - 'missing benchmark point after ingest', - ); - expect(isBenchmarkPointPurged(source.runId, source.runAttempt, purged)).toBe(true); - expect(() => assertRequiredPowerPointsRetained([], retained)).not.toThrow(); - } finally { - registry.splice(registry.indexOf(purged), 1); - } + it.each([undefined, 1, 3, '2'])( + 'rejects missing or incompatible manifest version %s', + (version) => { + const dir = fixture(); + changeManifest(dir, (manifest) => { + manifest['schema-version'] = version; + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('schema-version'); + }, + ); + it.each(['agentic_golden/gpu_metrics.csv', auditPath, benchmarkPath])( + 'rejects missing required artifact %s', + (file) => { + const dir = fixture(); + fs.unlinkSync(path.join(dir, file)); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('missing required artifact'); + }, + ); + it('rejects hash mismatches and explicit invalid evidence', () => { + const dir = fixture(); + fs.appendFileSync(path.join(dir, 'agentic_golden/gpu_metrics.csv'), '\n'); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('hash'); + const invalid = fixture(); + changeManifest(invalid, (manifest) => { + manifest.points[0].artifacts[0].validation_state = 'invalid'; + }); + expect(() => verifyRequiredPowerArtifacts(invalid, source)).toThrow('validation'); }); - - it('accepts retained aggregate copies but cannot replace a filtered required point with another identity', () => { - const requiredPoints = verifyRequiredPowerPublication(manifest(), artifacts(), source); - expect(() => - assertRequiredPowerPointsRetained(requiredPoints, [ - requiredPoints[0], - { ...requiredPoints[0] }, - ]), - ).not.toThrow(); - expect(() => - assertRequiredPowerPointsRetained(requiredPoints, [ - { ...requiredPoints[0], recipeFingerprint: 'other-recipe' }, - ]), - ).toThrow('missing benchmark point after ingest'); + it.each([undefined, null, 0, -1, '1000'])( + 'rejects missing, invalid and measured nonpositive energy separately: %s', + (energy) => { + const dir = fixture(); + changeArtifact(dir, benchmarkPath, (rows) => { + rows.total_gpu_energy_j = energy; + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow( + 'total_gpu_energy_j must be finite and positive', + ); + }, + ); + it('does not replace canonical AgentX users with a conflicting raw conc', () => { + const dir = fixture(); + changeArtifact(dir, benchmarkPath, (rows) => { + rows.conc = 1; + rows.users = 2; + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('missing benchmark point'); }); - - it('fails before ingestion when any declared concurrency, fingerprint or scenario is missing', () => { - for (const candidate of [ - [], - [{ ...row, conc: 2 }], - [{ ...row, recipe_fingerprint: '' }], - [{ ...row, isl: 1024 }], + it.each(['model', 'hardware', 'framework', 'precision'])( + 'rejects substituted %s identity', + (field) => { + const dir = fixture(); + changeManifest(dir, (manifest) => { + manifest.points[0].identity[field] = 'other'; + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow(`${field} identity differs`); + }, + ); + it('rejects absent physical GPU, duplicate physical GPU, and changed measurement boundaries', () => { + for (const edit of [ + (point: any) => { + point.devices = []; + }, + (point: any) => { + point.devices[0].gpu_uuid = '0'; + }, + (point: any) => { + point.measurement_window.end_time_unix += 1; + }, ]) { - expect(() => - verifyRequiredPowerPublication(manifest(), artifacts(candidate), source), - ).toThrow('missing benchmark point'); + const dir = fixture(); + changeManifest(dir, (manifest) => edit(manifest.points[0])); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('Required power:'); } }); - - it('does not retroactively require optional matrix points or legacy results', () => { - const scope = manifest([required, { ...required, conc: 2, 'require-power': false }]); - expect( - verifyRequiredPowerPublication(scope, artifacts([row, { conc: 2, power_valid: 0 }]), source), - ).toHaveLength(1); - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'required-power-')); - dirs.push(dir); - expect(verifyRequiredPowerArtifacts(dir, { ...source, headSha: null })).toHaveLength(0); + it('requires both prefill and decode physical evidence and matching role energy', () => { + const dir = multinodeFixture(); + expect(verifyRequiredPowerArtifacts(dir, source)).toHaveLength(1); + changeManifest(dir, (manifest) => { + manifest.points[0].devices.pop(); + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('missing physical GPU'); + const mismatched = multinodeFixture(); + changeArtifact(mismatched, benchmarkPath, (row) => { + row.prefill_gpu_energy_j = 100; + }); + expect(() => verifyRequiredPowerArtifacts(mismatched, source)).toThrow( + 'prefill energy differs', + ); }); - - it.each([ - { power_valid: 0 }, - { power_valid: '1' }, - { power_metric_schema_version: '2' }, - { avg_power_w: 0 }, - { avg_total_gpu_power_w: -1 }, - { total_gpu_energy_j: NaN }, - { joules_per_output_token: Infinity }, - { joules_per_output_token: undefined }, - { benchmark_outcome: { status: 'failed' } }, - ])('rejects invalid required measurements: %j', (change) => { - expect(() => - verifyRequiredPowerPublication(manifest(), artifacts([{ ...row, ...change }]), source), - ).toThrow('Required power:'); + it('normalizes producer dp-attn strings and binds nested planned role topology', () => { + const dir = fixture(); + changeArtifact(dir, benchmarkPath, (row) => { + row.dp_attention = 'false'; + }); + changeManifest(dir, (manifest) => { + manifest.points[0].topology.dp_attention = 'false'; + manifest.matrix.single_node.agentic[0]['dp-attn'] = false; + }); + expect(verifyRequiredPowerArtifacts(dir, source)).toHaveLength(1); + const split = multinodeFixture(); + changeManifest(split, (manifest) => { + manifest.matrix.multi_node.agentic[0].prefill = { tp: 8, 'num-worker': 1 }; + }); + expect(() => verifyRequiredPowerArtifacts(split, source)).toThrow( + 'prefill_tp topology differs from required matrix', + ); }); - - it('rejects conflicting copies and duplicate rows inside either source or aggregate file', () => { - expect(() => - verifyRequiredPowerPublication( - manifest(), - [...artifacts(), { path: 'results_bmk/agg.json', rows: [{ ...row, avg_power_w: 501 }] }], - source, - ), - ).toThrow('conflicting'); - expect(() => verifyRequiredPowerPublication(manifest(), artifacts([row, row]), source)).toThrow( - 'duplicate', + it('supports fixed-window sidecar names and AMD physical UUID artifacts', () => { + const dir = fixture(); + fs.renameSync( + path.join(dir, auditPath), + path.join(dir, 'agentic_golden/power_validation_conc1.json'), ); - expect(() => - verifyRequiredPowerPublication( - manifest(), - [...artifacts(), { path: 'results_bmk/agg.json', rows: [row, row] }], - source, - ), - ).toThrow('duplicate'); + fs.unlinkSync(path.join(dir, 'agentic_golden/gpu_metrics_identity.csv')); + const amdPath = 'agentic_golden/gpu_metrics_devices.json'; + write(dir, amdPath, { devices: [{ gpu: 0, uuid: 'amd-physical-uuid' }] }); + changeManifest(dir, (manifest) => { + const point = manifest.points[0]; + point.devices[0].gpu_uuid = 'amd-physical-uuid'; + point.artifacts = point.artifacts.filter( + (artifact: any) => !artifact.path.endsWith('gpu_metrics_identity.csv'), + ); + point.artifacts.find((artifact: any) => artifact.path === auditPath).path = + 'agentic_golden/power_validation_conc1.json'; + point.artifacts.push({ + path: amdPath, + sha256: createHash('sha256') + .update(fs.readFileSync(path.join(dir, amdPath))) + .digest('hex'), + validation_state: 'valid', + }); + }); + expect(verifyRequiredPowerArtifacts(dir, source)).toHaveLength(1); }); - - it('rejects stale source run, attempt or head, an empty scope, and duplicate planned points', () => { - for (const changed of [ - { runId: 999 }, - { runAttempt: 1 }, - { headSha: 'stale' }, - { headSha: null }, - ]) - expect(() => - verifyRequiredPowerPublication(manifest(), artifacts(), { ...source, ...changed }), - ).toThrow('manifest source'); - expect(() => verifyRequiredPowerPublication(manifest([]), [], source)).toThrow('no required'); - expect(() => - verifyRequiredPowerPublication(manifest([required, required]), artifacts(), source), - ).toThrow('duplicate matrix'); + it('rejects invalid sidecar verdict and device energy disagreement despite valid hashes', () => { + const dir = fixture(); + changeArtifact(dir, auditPath, (audit) => { + audit.power_valid = false; + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('invalid audit'); + const other = fixture(); + changeManifest(other, (manifest) => { + manifest.points[0].devices[0].energy_j = 2; + }); + expect(() => verifyRequiredPowerArtifacts(other, source)).toThrow('differs from audit'); }); - - it('reuses the declared scope from a successful earlier attempt of the same run and head', () => { - expect( - verifyRequiredPowerPublication({ ...manifest(), 'run-attempt': 1 }, artifacts(), source), - ).toHaveLength(1); + it('rejects duplicate points, required matrix omissions, wrong source, and paths outside the artifact root', () => { + const edits = [ + (manifest: any) => { + manifest.points.push(manifest.points[0]); + }, + (manifest: any) => { + manifest.matrix.single_node.agentic[0].conc = [1, 2]; + }, + (manifest: any) => { + manifest['run-id'] = 999; + }, + (manifest: any) => { + manifest.points[0].artifacts[0].path = '../secret'; + }, + ]; + for (const edit of edits) { + const dir = fixture(); + changeManifest(dir, edit); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('Required power:'); + } }); - - it('expands real ordinary AgentX planner concurrency arrays and fixed-sequence batches', () => { - // First B200 recipe from the ordinary Kimi K3 required-power planner, 2026-09-13. - const entry = agenticMatrix.multi_node.agentic[0]; - expect( - verifyRequiredPowerPublication( - { ...manifest(), matrix: agenticMatrix }, - artifacts([ - { - ...row, - infmax_model_prefix: 'kimik3', - hw: 'b200', - framework: 'dynamo-vllm', - precision: 'fp4', - recipe_fingerprint: entry['recipe-fingerprint'], - conc: entry.conc[0], - scenario_type: 'agentic-coding', - }, - ]), - source, - ), - ).toHaveLength(1); - const fixedMatrix = { - single_node: {}, - multi_node: { '8k1k': [{ ...required, conc: [1, 2, 4] }] }, - }; - expect( - verifyRequiredPowerPublication( - { ...manifest(), matrix: fixedMatrix }, - artifacts([1, 2, 4].map((conc) => ({ ...row, conc }))), - source, - ), - ).toHaveLength(3); - expect(() => - verifyRequiredPowerPublication({ ...manifest(), matrix: fixedMatrix }, artifacts(), source), - ).toThrow('missing benchmark'); + it('accepts identical aggregate copies but rejects conflicting duplicate rows', () => { + const dir = fixture(); + fs.mkdirSync(path.join(dir, 'results_bmk')); + fs.copyFileSync(path.join(dir, benchmarkPath), path.join(dir, 'results_bmk/agg.json')); + expect(verifyRequiredPowerArtifacts(dir, source)).toHaveLength(1); + const rows = json(dir, benchmarkPath); + rows.avg_power_w = 501; + write(dir, 'results_bmk/agg.json', rows); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('conflicting'); }); - - it('accepts AgentX and requires both energy roles only for a declared disaggregated recipe', () => { - const scope = manifest(); - const agentic = { ...required, 'scenario-type': 'agentic-coding', disagg: true }; - const agenticManifest = { - ...scope, - matrix: { single_node: {}, multi_node: { agentic: [agentic] } }, - }; - const split = { - ...row, - isl: undefined, - osl: undefined, - scenario_type: 'agentic-coding', - disagg: true, - prefill_gpu_energy_j: 3000, - decode_gpu_energy_j: 5000, - prefill_joules_per_input_token: 1, - decode_joules_per_output_token: 1.25, - }; - expect( - verifyRequiredPowerPublication(agenticManifest, artifacts([split]), source), - ).toHaveLength(1); + it('retains every required identity after local purges and backfills', () => { + const required = verifyRequiredPowerArtifacts(golden, source); expect(() => - verifyRequiredPowerPublication( - agenticManifest, - artifacts([{ ...split, decode_gpu_energy_j: undefined }]), - source, - ), - ).toThrow('decode_gpu_energy_j'); - const aggregate = { - ...agenticManifest, - matrix: { single_node: {}, multi_node: { agentic: [{ ...agentic, disagg: false }] } }, - }; - expect( - verifyRequiredPowerPublication( - aggregate, - artifacts([{ ...row, scenario_type: 'agentic-coding' }]), - source, - ), - ).toHaveLength(1); - }); - - it('reads the producer artifact and checks all rows before the ingest entry point writes', () => { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'required-power-')); - dirs.push(dir); - fs.mkdirSync(path.join(dir, 'required-power-sweep-manifest')); - fs.writeFileSync( - path.join(dir, 'required-power-sweep-manifest', 'sweep_manifest.json'), - JSON.stringify(manifest()), - ); - expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('missing benchmark'); - fs.mkdirSync(path.join(dir, 'bmk_qwen')); - fs.writeFileSync(path.join(dir, 'bmk_qwen', 'agg.json'), JSON.stringify(row)); - expect(verifyRequiredPowerArtifacts(dir, source)).toHaveLength(1); + assertRequiredPowerPointsRetained(required, [{ ...required[0], conc: 2 }]), + ).toThrow('missing benchmark point'); + expect(() => assertRequiredPowerPointsRetained(required, required)).not.toThrow(); }); }); diff --git a/packages/db/src/etl/required-power-publication.ts b/packages/db/src/etl/required-power-publication.ts index 170b3e926..d8f995b8a 100644 --- a/packages/db/src/etl/required-power-publication.ts +++ b/packages/db/src/etl/required-power-publication.ts @@ -1,9 +1,11 @@ import fs from 'node:fs'; +import { createHash } from 'node:crypto'; import path from 'node:path'; import { isDeepStrictEqual } from 'node:util'; import { mapBenchmarkRow, type BenchmarkParams } from './benchmark-mapper'; import { createSkipTracker } from './skip-tracker'; -import { REQUIRED_POWER_MANIFEST } from '../lib/ci-artifact-preparation'; +import { configCacheKey } from './config-cache'; +import { CHANGELOG_ARTIFACT_NAME, REQUIRED_POWER_MANIFEST } from '../lib/ci-artifact-preparation'; type JsonRow = Record; export interface RequiredPowerSource { @@ -14,6 +16,7 @@ export interface RequiredPowerSource { export interface BenchmarkArtifactRows { path: string; rows: unknown[]; + contents?: Buffer | string; } function object(value: unknown, label: string): JsonRow { @@ -45,11 +48,24 @@ export function assertRequiredPowerPointsRetained( required: readonly BenchmarkParams[], retained: readonly BenchmarkParams[], ): void { - const present = new Set(retained.map(mappedIdentity)); + const retainedIdentity = (row: BenchmarkParams) => + JSON.stringify([configCacheKey(row.config), row.offloadMode, mappedIdentity(row)]); + const present = new Map(retained.map((row) => [retainedIdentity(row), row])); for (const row of required) { - const key = mappedIdentity(row); + const key = retainedIdentity(row); if (!present.has(key)) throw new Error(`Required power: missing benchmark point after ingest ${key}`); + const actual = present.get(key)!; + for (const field of [ + 'power_valid', + 'power_metric_schema_version', + 'avg_power_w', + 'avg_total_gpu_power_w', + 'total_gpu_energy_j', + 'joules_per_output_token', + ]) + if (actual.metrics[field] !== row.metrics[field]) + throw new Error(`Required power: ${field} changed before ingest for ${key}`); } } @@ -60,20 +76,52 @@ export function verifyRequiredPowerPublication( source: RequiredPowerSource, ): BenchmarkParams[] { const manifest = object(manifestValue, 'sweep manifest'); + if (manifest['schema-version'] !== 2) + throw new Error('Required power: incompatible manifest schema-version (expected 2)'); + const publication = object(manifest.publication, 'publication policy'); + if ( + !['incremental', 'replacement'].includes(String(publication.mode)) || + !Array.isArray(publication.replacement_scope) || + (publication.mode === 'incremental' && publication.replacement_scope.length > 0) + ) + throw new Error('Required power: invalid publication policy'); + const replacementScopes = new Set(); + for (const value of publication.replacement_scope) { + const replacement = object(value, 'exact replacement scope'); + if ( + typeof replacement.curve_scope !== 'string' || + !replacement.curve_scope || + replacementScopes.has(replacement.curve_scope) || + !Number.isSafeInteger(replacement.previous_snapshot_workflow_run_id) || + Number(replacement.previous_snapshot_workflow_run_id) <= 0 || + !Array.isArray(replacement.removed_point_identities) || + replacement.removed_point_identities.length === 0 || + replacement.removed_point_identities.some((id) => typeof id !== 'string' || !id) || + new Set(replacement.removed_point_identities).size !== + replacement.removed_point_identities.length + ) + throw new Error('Required power: invalid exact replacement scope'); + replacementScopes.add(replacement.curve_scope); + } const declaredAttempt = manifest['run-attempt']; if ( manifest['run-id'] !== source.runId || + !Number.isSafeInteger(source.runId) || + source.runId <= 0 || + !Number.isSafeInteger(source.runAttempt) || + source.runAttempt <= 0 || typeof declaredAttempt !== 'number' || !Number.isSafeInteger(declaredAttempt) || declaredAttempt <= 0 || declaredAttempt > source.runAttempt || !source.headSha || + !/^[a-f0-9]{40}$/u.test(source.headSha) || manifest.head !== source.headSha ) throw new Error('Required power: manifest source run, attempt or head does not match'); const matrix = object(manifest.matrix, 'sweep matrix'); - const expected = new Map(); + const expected = new Map(); for (const topology of ['single_node', 'multi_node']) { const buckets = object(matrix[topology], topology); for (const [scenario, entries] of Object.entries(buckets)) { @@ -99,7 +147,7 @@ export function verifyRequiredPowerPublication( osl, ); if (expected.has(key)) throw new Error(`Required power: duplicate matrix point ${key}`); - expected.set(key, row.disagg === true); + expected.set(key, { ...row, matrixTopology: topology }); } } } @@ -140,7 +188,7 @@ export function verifyRequiredPowerPublication( 'total_gpu_energy_j', 'joules_per_output_token', ]; - if (expected.get(key)) + if (expected.get(key)?.disagg === true) fields.push( 'prefill_gpu_energy_j', 'decode_gpu_energy_j', @@ -158,6 +206,7 @@ export function verifyRequiredPowerPublication( for (const key of expected.keys()) { if (!seen.has(key)) throw new Error(`Required power: missing benchmark point ${key}`); } + verifyPointEvidence(manifest, expected, seen, artifacts); return [...seen.values()].map(({ point }) => point); } @@ -165,9 +214,21 @@ export function verifyRequiredPowerPublication( export function verifyRequiredPowerArtifacts( root: string, source: RequiredPowerSource, + required = false, ): BenchmarkParams[] { const manifestDir = path.join(root, REQUIRED_POWER_MANIFEST); - if (!fs.existsSync(manifestDir)) return []; + if (!fs.existsSync(manifestDir)) { + if (required) throw new Error('Required power: sweep manifest missing for required dispatch'); + const metadataDir = path.join(root, CHANGELOG_ARTIFACT_NAME); + if (fs.existsSync(metadataDir)) { + for (const name of fs.readdirSync(metadataDir).filter((file) => file.endsWith('.json'))) { + const metadata = JSON.parse(fs.readFileSync(path.join(metadataDir, name), 'utf8')); + if (metadata?.['require-power'] === true) + throw new Error('Required power: sweep manifest missing for required changelog scope'); + } + } + return []; + } const manifest = JSON.parse( fs.readFileSync(path.join(manifestDir, 'sweep_manifest.json'), 'utf8'), ); @@ -178,8 +239,28 @@ export function verifyRequiredPowerArtifacts( if (!fs.statSync(dir).isDirectory()) continue; for (const file of fs.readdirSync(dir)) { if (!file.endsWith('.json')) continue; - const data = JSON.parse(fs.readFileSync(path.join(dir, file), 'utf8')); - artifacts.push({ path: path.join(name, file), rows: Array.isArray(data) ? data : [data] }); + const contents = fs.readFileSync(path.join(dir, file)); + const data = JSON.parse(contents.toString('utf8')); + artifacts.push({ + path: path.join(name, file), + rows: Array.isArray(data) ? data : [data], + contents, + }); + } + } + for (const pointValue of Array.isArray(manifest.points) ? manifest.points : []) { + const point = object(pointValue, 'point'); + if (!Array.isArray(point.artifacts)) throw new Error('Required power: missing point artifacts'); + for (const value of point.artifacts) { + const artifact = object(value, 'artifact'); + const relative = safeArtifactPath(artifact.path); + const file = path.join(root, relative); + if (!fs.existsSync(file) || !fs.statSync(file).isFile()) + throw new Error(`Required power: missing required artifact ${relative}`); + if (!fs.realpathSync(file).startsWith(`${fs.realpathSync(root)}${path.sep}`)) + throw new Error(`Required power: artifact escapes bundle ${relative}`); + if (!artifacts.some((item) => item.path === relative)) + artifacts.push({ path: relative, rows: [], contents: fs.readFileSync(file) }); } } const points = verifyRequiredPowerPublication(manifest, artifacts, source); @@ -189,3 +270,424 @@ export function verifyRequiredPowerArtifacts( ); return points; } + +function digest(artifact: BenchmarkArtifactRows): string { + return createHash('sha256').update(artifact.contents!).digest('hex'); +} + +function safeArtifactPath(value: unknown): string { + if ( + typeof value !== 'string' || + !value || + value.includes('\\') || + path.posix.isAbsolute(value) || + value.split('/').some((part) => !part || part === '.' || part === '..') + ) + throw new Error('Required power: invalid artifact path'); + return value; +} + +function positive(value: unknown, label: string): number { + if (typeof value !== 'number' || !Number.isFinite(value) || value <= 0) + throw new Error(`Required power: ${label} must be finite and positive`); + return value; +} + +function nonempty(value: unknown, label: string): string { + if (typeof value !== 'string' || !value.trim()) + throw new Error(`Required power: missing ${label}`); + return value; +} + +function verifyPointEvidence( + manifest: JsonRow, + expected: Map, + seen: Map, + artifacts: readonly BenchmarkArtifactRows[], +): void { + if (!Array.isArray(manifest.points) || manifest.points.length !== expected.size) + throw new Error('Required power: point manifest does not cover the exact required matrix'); + const verified = new Set(); + const files = new Map(artifacts.map((artifact) => [artifact.path, artifact])); + for (const declaredValue of manifest.points) { + const declared = object(declaredValue, 'point'); + const id = object(declared.identity, 'point identity'); + const key = identity( + id.recipe_fingerprint, + id.concurrency, + String(id.benchmark_type), + id.isl, + id.osl, + ); + const matched = seen.get(key); + const matrix = expected.get(key); + if (!matched || !matrix || verified.has(key)) + throw new Error(`Required power: unexpected or duplicate manifest point ${key}`); + verified.add(key); + if (nonempty(declared.config_key, 'config_key') !== matrix['exp-name']) + throw new Error(`Required power: config_key differs from required matrix for ${key}`); + const { row, point } = matched; + for (const [field, rawField] of [ + ['model', 'infmax_model_prefix'], + ['hardware', 'hw'], + ['framework', 'framework'], + ['precision', 'precision'], + ]) { + if ( + nonempty(id[field], field) !== + (row[rawField] ?? (field === 'model' ? row.model : undefined)) + ) + throw new Error(`Required power: ${field} identity differs from benchmark for ${key}`); + } + for (const [field, matrixField] of [ + ['model', 'model-prefix'], + ['hardware', 'runner'], + ['framework', 'framework'], + ['precision', 'precision'], + ]) { + if (id[field] !== matrix[matrixField]) + throw new Error( + `Required power: ${field} identity differs from required matrix for ${key}`, + ); + } + const topology = object(declared.topology, 'point topology'); + if ( + topology.disagg !== point.config.disagg || + topology.is_multinode !== point.config.isMultinode || + topology.disagg !== (matrix.disagg === true) || + topology.is_multinode !== (matrix.matrixTopology === 'multi_node') + ) + throw new Error(`Required power: topology differs from benchmark or matrix for ${key}`); + const rawGpuCount = + row.num_gpus ?? + (topology.is_multinode + ? Number(row.num_prefill_gpu) + Number(row.num_decode_gpu) + : Number(row.tp) * Number(row.pp ?? 1) * Number(row.pcp_size ?? 1)); + if (topology.num_gpus !== rawGpuCount) + throw new Error(`Required power: num_gpus topology differs from benchmark for ${key}`); + const topologyFields = [ + 'num_prefill_gpu', + 'num_decode_gpu', + 'tp', + 'ep', + 'dp_attention', + 'prefill_tp', + 'prefill_ep', + 'prefill_num_workers', + 'decode_tp', + 'decode_ep', + 'decode_num_workers', + 'pp', + 'pcp_size', + 'dcp_size', + 'prefill_pp', + 'decode_pp', + 'prefill_pcp_size', + 'decode_pcp_size', + 'prefill_dcp_size', + 'decode_dcp_size', + 'prefill_dp_attention', + 'decode_dp_attention', + ]; + for (const field of topologyFields) { + if (topology[field] !== row[field]) + throw new Error(`Required power: ${field} topology differs from benchmark for ${key}`); + const role = field.startsWith('prefill_') + ? 'prefill' + : field.startsWith('decode_') + ? 'decode' + : null; + const roleField = role ? field.slice(role.length + 1) : field; + const matrixField = + roleField === 'num_workers' + ? 'num-worker' + : roleField === 'dp_attention' + ? 'dp-attn' + : roleField.replaceAll('_', '-'); + let planned = role + ? matrix[role] + ? object(matrix[role], `${role} matrix`)[matrixField] + : matrix[field.replaceAll('_', '-')] + : matrix[matrixField]; + if ( + role === 'decode' && + matrix.decode && + object(matrix.decode, 'decode matrix')['num-worker'] === 0 && + ['tp', 'ep', 'pp', 'pcp_size', 'dcp_size'].includes(roleField) + ) + planned = ['tp', 'ep'].includes(roleField) ? 0 : 1; + let actual = + topology[field] ?? + (['pp', 'pcp_size', 'dcp_size'].includes(roleField) + ? 1 + : roleField === 'dp_attention' + ? false + : undefined); + if (roleField === 'dp_attention' && typeof actual === 'string') actual = actual === 'true'; + const normalizedPlanned = + roleField === 'dp_attention' && typeof planned === 'string' ? planned === 'true' : planned; + if (planned !== undefined && actual !== normalizedPlanned) + throw new Error( + `Required power: ${field} topology differs from required matrix for ${key}`, + ); + } + let plannedGpuCount = matrix['num-gpus']; + if (matrix.prefill && matrix.decode) { + plannedGpuCount = 0; + for (const role of ['prefill', 'decode']) { + const planned = object(matrix[role], `${role} matrix`); + const count = + Number(planned.tp) * + Number(planned.pp ?? 1) * + Number(planned['pcp-size'] ?? 1) * + Number(planned['num-worker']); + if (count !== topology[`num_${role}_gpu`]) + throw new Error( + `Required power: ${role} GPU count differs from required matrix for ${key}`, + ); + plannedGpuCount = Number(plannedGpuCount) + count; + } + } else if (plannedGpuCount === undefined && matrix.tp !== undefined) { + plannedGpuCount = + Number(matrix.tp) * Number(matrix.pp ?? 1) * Number(matrix['pcp-size'] ?? 1); + } + if (plannedGpuCount !== undefined && plannedGpuCount !== topology.num_gpus) + throw new Error(`Required power: physical GPU count differs from required matrix for ${key}`); + const expectedCount = positive(topology.num_gpus, 'topology num_gpus'); + if (!Number.isSafeInteger(expectedCount)) + throw new Error('Required power: invalid physical GPU count'); + const window = object(declared.measurement_window, 'measurement window'); + const start = positive(window.start_time_unix, 'window start'); + const end = positive(window.end_time_unix, 'window end'); + if (end <= start) throw new Error('Required power: invalid measurement window boundaries'); + if (!Array.isArray(declared.artifacts) || declared.artifacts.length === 0) + throw new Error('Required power: missing required artifacts'); + const evidence = new Map(); + for (const value of declared.artifacts) { + const artifact = object(value, 'required artifact'); + const relative = safeArtifactPath(artifact.path); + if (evidence.has(relative)) + throw new Error(`Required power: duplicate required artifact ${relative}`); + const file = files.get(relative); + if (!file || file.contents === undefined || file.contents.length === 0) + throw new Error(`Required power: missing required artifact ${relative}`); + if ( + artifact.validation_state !== 'valid' || + typeof artifact.sha256 !== 'string' || + !/^[a-f0-9]{64}$/u.test(artifact.sha256) || + createHash('sha256').update(file.contents).digest('hex') !== artifact.sha256 + ) + throw new Error(`Required power: invalid artifact validation or hash ${relative}`); + evidence.set(relative, file); + } + if ( + ![...evidence.values()].some((file) => + file.rows.some((candidate) => isDeepStrictEqual(candidate, row)), + ) + ) + throw new Error(`Required power: benchmark artifact is not hash-bound for ${key}`); + const byName = (name: string): BenchmarkArtifactRows => { + const matches = [...evidence.values()].filter( + (file) => path.posix.basename(file.path) === name, + ); + if (matches.length !== 1) + throw new Error(`Required power: expected one required ${name} artifact`); + return matches[0]; + }; + const sidecars = [...evidence.values()].filter((file) => + /^power_validation.*\.json$/u.test(path.posix.basename(file.path)), + ); + if (sidecars.length !== 1) + throw new Error('Required power: expected one required power validation artifact'); + const audit = object(JSON.parse(sidecars[0].contents!.toString()), 'power validation'); + const auditWindow = object(audit.benchmark_window, 'audit benchmark window'); + if ( + audit.power_valid !== true || + auditWindow.start_time_unix !== start || + auditWindow.end_time_unix !== end || + audit.expected_gpu_count !== expectedCount || + audit.observed_gpu_count !== expectedCount + ) + throw new Error( + `Required power: invalid audit verdict, window or physical GPU coverage for ${key}`, + ); + if (topology.is_multinode && audit.telemetry_kind === 'native_multinode_smi') { + if (!Array.isArray(audit.nodes) || audit.nodes.length === 0) + throw new Error('Required power: missing native node receipts'); + const traces = [...evidence.values()].filter( + (file) => path.posix.basename(file.path) === 'gpu_metrics.csv', + ); + if (traces.length !== audit.nodes.length) + throw new Error('Required power: missing native node telemetry'); + for (const file of traces) { + const directory = path.posix.dirname(file.path); + const manifestFile = evidence.get(`${directory}/manifest.json`); + if (!manifestFile) throw new Error('Required power: missing native node manifest'); + const nodeManifest = object( + JSON.parse(manifestFile.contents!.toString()), + 'native node manifest', + ); + if (nodeManifest.lifecycle !== 'complete' || nodeManifest.collector_exit_code !== 0) + throw new Error('Required power: invalid native node collection'); + const receipt = (audit.nodes as JsonRow[]).find((node) => node.node === nodeManifest.node); + if ( + !receipt || + receipt.manifest_sha256 !== digest(manifestFile) || + receipt.telemetry_sha256 !== digest(file) + ) + throw new Error('Required power: native node receipt differs from evidence'); + for (const hash of [receipt.identity_sha256, receipt.identity_end_sha256]) + if ( + ![...evidence.values()].some( + (item) => path.posix.dirname(item.path) === directory && digest(item) === hash, + ) + ) + throw new Error('Required power: missing native physical identity evidence'); + } + } else if (topology.is_multinode) { + byName('samples.csv'); + byName('manifest.json'); + if ( + ![...evidence.keys()].some((file) => file.includes('/windows/') && file.endsWith('.json')) + ) + throw new Error('Required power: missing central measurement window artifact'); + } else byName('gpu_metrics.csv'); + const energies = object(audit.per_gpu_energy_j, 'audit device energy'); + let auditDevices: JsonRow[]; + if (audit.per_gpu_role === undefined) { + const nodeFiles = [...evidence.values()].filter((file) => + ['power_node.txt', 'gpu_metrics_node.txt'].includes(path.posix.basename(file.path)), + ); + if (nodeFiles.length !== 1) + throw new Error('Required power: expected one node identity artifact'); + const node = nodeFiles[0].contents!.toString().trim(); + const identityFiles = [...evidence.values()].filter((file) => + /^(?:gpu_metrics_identity\.csv|gpu_metrics_devices\.json)$/u.test( + path.posix.basename(file.path), + ), + ); + if (identityFiles.length !== 1) + throw new Error('Required power: expected one physical GPU identity artifact'); + const uuidRows: [string, string][] = []; + if (identityFiles[0].path.endsWith('.csv')) { + const lines = identityFiles[0].contents!.toString().trim().split(/\r?\n/u); + const columns = lines + .shift()! + .split(',') + .map((part) => part.trim().replaceAll('"', '').toLowerCase()); + const indexColumn = columns.indexOf('index'); + const uuidColumn = columns.indexOf('uuid'); + if (indexColumn === -1 || uuidColumn === -1) + throw new Error('Required power: invalid physical GPU identity CSV'); + for (const line of lines) { + const fields = line.split(',').map((part) => part.trim().replaceAll('"', '')); + uuidRows.push([fields[indexColumn], fields[uuidColumn]]); + } + } else { + const visit = (value: unknown): void => { + if (Array.isArray(value)) value.forEach(visit); + else if (value && typeof value === 'object') { + const deviceRow = Object.fromEntries( + Object.entries(value).map(([field, item]) => [field.toLowerCase(), item]), + ); + if ('gpu' in deviceRow && 'uuid' in deviceRow) + uuidRows.push([String(deviceRow.gpu), String(deviceRow.uuid)]); + else Object.values(value).forEach(visit); + } + }; + visit(JSON.parse(identityFiles[0].contents!.toString())); + } + if ( + uuidRows.length === 0 || + uuidRows.some( + ([index, uuid]) => + !/^\d+$/u.test(index) || !uuid || ['n/a', 'none', 'null'].includes(uuid.toLowerCase()), + ) || + new Set(uuidRows.map(([index]) => index)).size !== uuidRows.length || + new Set(uuidRows.map(([, uuid]) => uuid)).size !== uuidRows.length + ) + throw new Error('Required power: invalid or duplicate physical GPU identity'); + const uuids = new Map(uuidRows); + auditDevices = Object.entries(energies).map(([index, energy]) => ({ + node, + gpu_uuid: uuids.get(index), + role: 'aggregate', + energy_j: energy, + })); + } else { + const roles = object(audit.per_gpu_role, 'audit device roles'); + const nativeNodes = new Map(); + if (audit.telemetry_kind === 'native_multinode_smi') + for (const value of audit.nodes as unknown[]) { + const receipt = object(value, 'native node receipt'); + for (const uuid of Object.values( + object(receipt.physical_gpu_ids, 'native GPU identities'), + )) { + const uuidValue = nonempty(uuid, 'native GPU UUID'); + if (nativeNodes.has(uuidValue)) + throw new Error('Required power: duplicate native GPU identity'); + nativeNodes.set(uuidValue, nonempty(receipt.node, 'native node')); + } + } + auditDevices = Object.entries(energies).map(([device, energy]) => { + const slash = device.lastIndexOf('/'); + return { + node: nativeNodes.get(device) ?? device.slice(0, slash), + gpu_uuid: nativeNodes.has(device) ? device : device.slice(slash + 1), + role: roles[device] === 'agg' ? 'aggregate' : roles[device], + energy_j: energy, + }; + }); + } + if ( + !Array.isArray(declared.devices) || + declared.devices.length !== expectedCount || + auditDevices.length !== expectedCount + ) + throw new Error(`Required power: missing physical GPU evidence for ${key}`); + const devices = new Set(); + const roles = new Map(); + let totalEnergy = 0; + const roleEnergy = new Map(); + const nodes = new Set(); + for (const value of declared.devices) { + const device = object(value, 'physical GPU'); + const node = nonempty(device.node, 'GPU node'); + nodes.add(node); + const uuid = nonempty(device.gpu_uuid, 'physical GPU UUID'); + if (['n/a', 'none', 'null'].includes(uuid.toLowerCase()) || devices.has(uuid)) + throw new Error('Required power: duplicate or invalid physical GPU UUID'); + devices.add(uuid); + if (!['aggregate', 'prefill', 'decode'].includes(String(device.role))) + throw new Error('Required power: invalid physical GPU role'); + roles.set(String(device.role), (roles.get(String(device.role)) ?? 0) + 1); + const energy = positive(device.energy_j, 'device energy_j'); + totalEnergy += energy; + roleEnergy.set(String(device.role), (roleEnergy.get(String(device.role)) ?? 0) + energy); + if (!auditDevices.some((actual) => isDeepStrictEqual(actual, device))) + throw new Error(`Required power: physical GPU evidence differs from audit for ${key}`); + } + if (nodes.size !== (matrix['node-count'] ?? 1)) + throw new Error( + `Required power: participating node count differs from required matrix for ${key}`, + ); + if (topology.disagg) { + if ( + roles.get('prefill') !== positive(topology.num_prefill_gpu, 'prefill GPU count') || + roles.get('decode') !== positive(topology.num_decode_gpu, 'decode GPU count') || + roles.has('aggregate') + ) + throw new Error(`Required power: missing prefill or decode evidence for ${key}`); + for (const role of ['prefill', 'decode']) { + const energy = positive(row[`${role}_gpu_energy_j`], `${role} energy`); + if (Math.abs((roleEnergy.get(role) ?? 0) - energy) > Math.max(0.01, energy * 1e-4)) + throw new Error(`Required power: ${role} energy differs from device evidence for ${key}`); + } + } else if (roles.get('aggregate') !== expectedCount) { + throw new Error(`Required power: invalid aggregate GPU roles for ${key}`); + } + const energy = positive(row.total_gpu_energy_j, 'total_gpu_energy_j'); + if (Math.abs(totalEnergy - energy) > Math.max(0.01, energy * 1e-4)) + throw new Error(`Required power: total GPU energy differs from device evidence for ${key}`); + } +} diff --git a/packages/db/src/ingest-ci-run.ts b/packages/db/src/ingest-ci-run.ts index 932a6f194..debb9f463 100644 --- a/packages/db/src/ingest-ci-run.ts +++ b/packages/db/src/ingest-ci-run.ts @@ -60,6 +60,7 @@ import { readReusedIngestMetadata, } from './etl/reused-ingest-metadata'; import { mapBenchmarkRow, type BenchmarkParams } from './etl/benchmark-mapper'; +import { preflightRequiredPowerCurves } from './etl/required-power-curve'; import { assertRequiredPowerPointsRetained, verifyRequiredPowerArtifacts, @@ -335,11 +336,15 @@ async function main(): Promise { } } - const requiredPowerPoints = verifyRequiredPowerArtifacts(artifactsDir, { - runId, - runAttempt: runAttemptNum, - headSha: ghInfo?.headSha ?? null, - }); + const requiredPowerPoints = verifyRequiredPowerArtifacts( + artifactsDir, + { + runId, + runAttempt: runAttemptNum, + headSha: ghInfo?.headSha ?? null, + }, + process.env.INGEST_REQUIRE_POWER === 'true', + ); if (requiredPowerPoints.length > 0) console.log(` Required power: ${requiredPowerPoints.length} source benchmark points verified`); @@ -411,6 +416,18 @@ async function main(): Promise { if (evalsOnly && requiredPowerPoints.length > 0) throw new Error('Required power: benchmark scope cannot be published as an evals-only run'); + if (requiredPowerPoints.length > 0) + await preflightRequiredPowerCurves( + sql, + artifactsDir, + { + runId, + runAttempt: runAttemptNum, + headSha: ghInfo?.headSha ?? null, + }, + { date, runStartedAt: workflowGhInfo?.runStartedAt ?? null, appendOnly }, + ); + const workflowRunId = await getOrCreateWorkflowRun({ githubRunId: runId, runAttempt: runAttemptNum, diff --git a/packages/db/src/preflight-power-publication.ts b/packages/db/src/preflight-power-publication.ts new file mode 100644 index 000000000..031c07170 --- /dev/null +++ b/packages/db/src/preflight-power-publication.ts @@ -0,0 +1,36 @@ +import { verifyRequiredPowerArtifacts } from './etl/required-power-publication'; + +// Deliberately artifact-only: this entry point opens no DB connection and writes no files. +const [root, runId, runAttempt, headSha, option] = process.argv.slice(2); +if ( + !root || + !runId || + !runAttempt || + !/^\d+$/u.test(runId) || + !/^\d+$/u.test(runAttempt) || + !Number.isSafeInteger(Number(runId)) || + !Number.isSafeInteger(Number(runAttempt)) || + Number(runId) <= 0 || + Number(runAttempt) <= 0 || + !headSha || + !/^[a-f0-9]{40}$/u.test(headSha) || + (option !== undefined && option !== '--optional-power') || + process.argv.length > 7 +) + throw new Error( + 'Usage: preflight-power-publication.ts [--optional-power]', + ); + +const points = verifyRequiredPowerArtifacts( + root, + { runId: Number(runId), runAttempt: Number(runAttempt), headSha }, + option !== '--optional-power', +); +console.log( + JSON.stringify({ + status: 'validated_artifacts', + requiredPoints: points.length, + databaseWrites: 0, + curvePreservation: 'not_checked_requires_published_state', + }), +); diff --git a/packages/db/src/prepare-ci-artifacts.test.ts b/packages/db/src/prepare-ci-artifacts.test.ts new file mode 100644 index 000000000..0f444791b --- /dev/null +++ b/packages/db/src/prepare-ci-artifacts.test.ts @@ -0,0 +1,90 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; + +const fixture = fileURLToPath( + new URL('../../../docs/fixtures/powerx-manifest-v2/artifacts/', import.meta.url), +); +const mocks = vi.hoisted(() => ({ names: [] as string[], fixture: '', corrupt: false })); +vi.mock('node:child_process', () => ({ + execFileSync: vi.fn(() => JSON.stringify({ head_sha: 'b'.repeat(40), run_attempt: 1 })), +})); +vi.mock('./lib/github-artifacts.js', () => ({ + listRunArtifacts: () => + mocks.names.map((name, id) => ({ + id, + name, + expired: false, + created_at: '2026-09-16T00:00:00Z', + })), + downloadArtifact: (artifact: { name: string }, root: string) => { + fs.cpSync(path.join(mocks.fixture, artifact.name), path.join(root, artifact.name), { + recursive: true, + }); + if (mocks.corrupt && artifact.name === 'agentic_golden') + fs.appendFileSync(path.join(root, artifact.name, 'gpu_metrics.csv'), 'corrupt\n'); + }, +})); + +let directory: string; +let originalArgs: string[]; +let originalExitCode: typeof process.exitCode; +beforeEach(() => { + vi.resetModules(); + directory = fs.mkdtempSync(path.join(os.tmpdir(), 'power-prepare-')); + mocks.fixture = fixture; + mocks.names = fs.readdirSync(fixture); + mocks.corrupt = false; + originalArgs = process.argv; + originalExitCode = process.exitCode; + process.argv = ['bun', 'prepare-ci-artifacts.ts']; + process.exitCode = undefined; + vi.stubEnv('SOURCE_RUN_ID', '123'); + vi.stubEnv('MERGE_RUN_ID', '123'); + vi.stubEnv('ARTIFACTS_PATH', directory); + vi.stubEnv('INGEST_REQUIRE_POWER', 'true'); + vi.stubEnv('GITHUB_OUTPUT', ''); + vi.spyOn(console, 'log').mockImplementation(() => {}); + vi.spyOn(console, 'error').mockImplementation(() => {}); +}); +afterEach(() => { + process.argv = originalArgs; + process.exitCode = originalExitCode; + vi.unstubAllEnvs(); + vi.restoreAllMocks(); + fs.rmSync(directory, { recursive: true, force: true }); +}); + +describe('artifact preparation gate before migrations', () => { + it('accepts the actual golden source bundle without a database connection', async () => { + await import('./prepare-ci-artifacts'); + expect(process.exitCode).toBeUndefined(); + expect(console.error).not.toHaveBeenCalled(); + }); + + it('fails the preparation command when both manifest and changelog are lost', async () => { + mocks.names = mocks.names.filter( + (name) => name !== 'required-power-sweep-manifest' && name !== 'changelog-metadata', + ); + await import('./prepare-ci-artifacts'); + expect(process.exitCode).toBe(1); + expect(console.error).toHaveBeenCalledWith(expect.stringContaining('required dispatch')); + }); + + it('fails preparation on corrupt downloaded telemetry', async () => { + mocks.corrupt = true; + await import('./prepare-ci-artifacts'); + expect(process.exitCode).toBe(1); + expect(console.error).toHaveBeenCalledWith(expect.stringContaining('hash')); + }); + + it('retains optional-power compatibility when there is no required marker', async () => { + vi.stubEnv('INGEST_REQUIRE_POWER', 'false'); + mocks.names = ['bmk_agentic_golden']; + await import('./prepare-ci-artifacts'); + expect(process.exitCode).toBeUndefined(); + expect(console.error).not.toHaveBeenCalled(); + }); +}); diff --git a/packages/db/src/prepare-ci-artifacts.ts b/packages/db/src/prepare-ci-artifacts.ts index fbf834d70..f0c35780e 100644 --- a/packages/db/src/prepare-ci-artifacts.ts +++ b/packages/db/src/prepare-ci-artifacts.ts @@ -6,6 +6,7 @@ import path from 'node:path'; import { buildArtifactPlan } from './lib/ci-artifact-preparation.js'; import { downloadArtifact, listRunArtifacts, type ArtifactMeta } from './lib/github-artifacts.js'; +import { verifyRequiredPowerArtifacts } from './etl/required-power-publication.js'; const DEFAULT_REPO = 'SemiAnalysisAI/InferenceX'; @@ -148,6 +149,16 @@ function main(): void { console.log(`Downloading artifact: ${artifact.name}`); downloadWithRetries(artifact, artifactsPath); } + // This command precedes migrations: malformed required input must not reach any DB write. + verifyRequiredPowerArtifacts( + artifactsPath, + { + runId: Number(sourceRunId), + runAttempt: sourceMetadata.run_attempt ?? 1, + headSha: sourceMetadata.head_sha ?? null, + }, + process.env.INGEST_REQUIRE_POWER === 'true', + ); if (plan.reused) { writeReuseMetadata(artifactsPath, sourceRunId, mergeRunId, sourceMetadata, mergeMetadata); } diff --git a/packages/db/src/verify-power-publication.ts b/packages/db/src/verify-power-publication.ts index 84bf86723..fbd34ab68 100644 --- a/packages/db/src/verify-power-publication.ts +++ b/packages/db/src/verify-power-publication.ts @@ -33,7 +33,7 @@ try { join workflow_runs wr on wr.id = br.workflow_run_id where wr.github_run_id = ${manifest.runId} and wr.run_attempt = ${manifest.runAttempt} and (br.benchmark_type = 'agentic_traces' or - (br.benchmark_type = 'single_turn' and br.isl = 8192 and br.osl = 1024)) + (br.benchmark_type = 'single_turn' and br.isl in (1024, 8192) and br.osl = 1024)) `; const errors = [ ...(manifest.ingestErrors ?? []), From ca4f61953ea73f2fc9417a62299d5efaa8582ad3 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sun, 20 Sep 2026 21:04:32 -0700 Subject: [PATCH 015/103] fix: delete purged telemetry explicitly and report what went MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Migration 016 declares `on delete cascade` from gpu_metric_series to workflow_runs and from benchmark_result_gpu_metrics to benchmark_results, so a purge already removed telemetry — with no count in the preview and no line in the transcript. Past GitHub's 90-day artifact retention the stored samples are the only copy, which is the reason 016 exists, so that loss should not be invisible at the moment the operator confirms it. The same rows are removed as before; this is accounting, not a change of behaviour. previewPurge now prints the series and sample counts alongside the benchmark and server_log counts, a whole-run purge deletes the series itself before dropping workflow_runs and logs what went, and a point purge drops only the point-to-series links. The series stays with its run in that second case: /api/gpu-metrics?runId= reads series by run rather than through the links, and other points of the same run may still reference it. New lib/telemetry-purge.ts holds the three queries, covered by 11 PGlite tests against the real 001 and 016 migrations so the cascades under test are the ones the migration declares. 中文:迁移 016 声明了 gpu_metric_series 到 workflow_runs、 benchmark_result_gpu_metrics 到 benchmark_results 的 `on delete cascade`,因此 purge 一直在删除遥测数据,但预览里没有计数、日志里没有记录。超过 GitHub 90 天 产物保留期后,库里的采样就是唯一副本,这正是 016 存在的理由,所以操作者确认的 那一刻不应该看不见这笔损失。 删除的行与此前完全相同,这次改的是账目而不是行为。previewPurge 现在会在 benchmark 与 server_log 计数旁一并打印 series 与采样数;整个 run 的 purge 会在 删除 workflow_runs 之前先显式删除 series 并记录;单点 purge 只解除点与 series 的 链接。后一种情况下 series 会随 run 保留,因为 /api/gpu-metrics?runId= 是按 run 读取 series 而不是走这些链接,而且同一个 run 的其他点可能仍在引用它。 新增的 lib/telemetry-purge.ts 收拢这三条查询,由 11 个 PGlite 测试覆盖,直接对 真实的 001 与 016 迁移运行,确保被测的正是迁移声明的那些级联。 --- packages/db/src/apply-overrides.ts | 28 +++- packages/db/src/lib/telemetry-purge.test.ts | 160 ++++++++++++++++++++ packages/db/src/lib/telemetry-purge.ts | 100 ++++++++++++ 3 files changed, 287 insertions(+), 1 deletion(-) create mode 100644 packages/db/src/lib/telemetry-purge.test.ts create mode 100644 packages/db/src/lib/telemetry-purge.ts diff --git a/packages/db/src/apply-overrides.ts b/packages/db/src/apply-overrides.ts index 6ac41680b..62188b705 100644 --- a/packages/db/src/apply-overrides.ts +++ b/packages/db/src/apply-overrides.ts @@ -26,6 +26,13 @@ import { type Sql, createAdminSql, refreshLatestBenchmarks } from './etl/db-util import { jsonbParam } from './lib/backfill-runner.js'; import { planBenchmarkPointBackfill } from './lib/benchmark-point-backfill.js'; import { selectRunOverrides } from './lib/run-override-selection.js'; +import { + type TelemetryPurgeCounts, + countRunTelemetry, + deleteRunTelemetry, + describeTelemetry, + unlinkPointTelemetry, +} from './lib/telemetry-purge.js'; import { type BenchmarkPointBackfill, type ChangelogBackfill, @@ -315,6 +322,7 @@ interface PurgeTarget { stats: number; evals: number; changelogs: number; + telemetry: TelemetryPurgeCounts; } /** @@ -358,17 +366,21 @@ async function previewPurge( ); } - const [[bmk], [stats], [evals], [changelogs], [logs]] = await Promise.all([ + const [[bmk], [stats], [evals], [changelogs], [logs], telemetry] = await Promise.all([ sql`SELECT count(*)::int AS n FROM benchmark_results WHERE workflow_run_id = ANY(${wrIds})`, sql`SELECT count(*)::int AS n FROM run_stats WHERE workflow_run_id = ANY(${wrIds})`, sql`SELECT count(*)::int AS n FROM eval_results WHERE workflow_run_id = ANY(${wrIds})`, sql`SELECT count(*)::int AS n FROM changelog_entries WHERE workflow_run_id = ANY(${wrIds})`, sql`SELECT count(DISTINCT server_log_id)::int AS n FROM benchmark_results WHERE workflow_run_id = ANY(${wrIds}) AND server_log_id IS NOT NULL`, + countRunTelemetry(sql, wrIds), ]); console.log( ` ${bmk.n} benchmarks, ${logs.n} server_logs, ${stats.n} run_stats, ${evals.n} evals, ${changelogs.n} changelogs`, ); + // Surfaced separately: past GitHub's 90-day artifact retention these samples + // are the only copy, so the operator should see them before confirming. + if (telemetry.series > 0) console.log(` ${describeTelemetry(telemetry)}`); return { githubRunId, @@ -378,6 +390,7 @@ async function previewPurge( stats: stats.n, evals: evals.n, changelogs: changelogs.n, + telemetry, }; } @@ -404,6 +417,14 @@ async function purgeBenchmarkResults(tx: Sql, resultIds: number[]): Promise 0) + console.log(` unlinked ${unlinked} gpu_metric_series link(s); series kept with the run.`); + await tx`DELETE FROM benchmark_results WHERE id = ANY(${resultIds})`; const sIds = logRows.map((r) => r.id as number); @@ -493,6 +514,11 @@ async function purge(wrIds: number[]): Promise { await tx`DELETE FROM eval_results WHERE workflow_run_id = ANY(${wrIds})`; await tx`DELETE FROM changelog_entries WHERE workflow_run_id = ANY(${wrIds})`; + // Telemetry too. The workflow_runs delete below would cascade it away anyway, + // but silently; deleting it here reports the cost of an irreversible loss. + const telemetry = await deleteRunTelemetry(tx, wrIds); + if (telemetry.series > 0) console.log(` deleted ${describeTelemetry(telemetry)}.`); + // Parent last (target the specific workflow_runs rows so partial purges // leave sibling attempts of the same github_run_id intact) await tx`DELETE FROM workflow_runs WHERE id = ANY(${wrIds})`; diff --git a/packages/db/src/lib/telemetry-purge.test.ts b/packages/db/src/lib/telemetry-purge.test.ts new file mode 100644 index 000000000..9224172f3 --- /dev/null +++ b/packages/db/src/lib/telemetry-purge.test.ts @@ -0,0 +1,160 @@ +/** + * Telemetry purge behaviour against a PGlite database with the real migrations + * applied, so the FK cascades declared in 016_gpu_metrics.sql are the ones under + * test rather than a hand-written stand-in. + * + * The regression these cover: before this change both purge levels relied on + * `on delete cascade`, so a purge destroyed telemetry with no count anywhere. The + * point-level case was worse than silent — it left the series stranded, paying + * storage while the per-point endpoint could never reach it again. + */ + +import fs from 'node:fs'; + +import { PGlite } from '@electric-sql/pglite'; +import type postgres from 'postgres'; +import { afterAll, beforeAll, beforeEach, describe, expect, it } from 'vitest'; + +import { + countRunTelemetry, + deleteRunTelemetry, + describeTelemetry, + unlinkPointTelemetry, +} from './telemetry-purge'; + +type Sql = postgres.Sql; +let db: PGlite; +let sql: Sql; + +function queryClient(database: Pick) { + const client = async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await database.query(query, values); + return result.rows; + }; + return Object.assign(client, { json: JSON.stringify, array: (value: unknown) => value }); +} + +const count = async (table: string): Promise => { + const result = await db.query<{ n: number }>(`SELECT count(*)::int AS n FROM ${table}`); + return result.rows[0].n; +}; + +beforeAll(async () => { + db = await PGlite.create(); + for (const name of ['001_initial_schema.sql', '016_gpu_metrics.sql']) { + await db.exec(fs.readFileSync(new URL(`../../migrations/${name}`, import.meta.url), 'utf8')); + } + sql = queryClient(db) as unknown as Sql; +}, 20_000); + +afterAll(async () => { + await db?.close(); +}); + +/** + * Two runs. Run 1 has two benchmark points sharing one series plus a second + * series linked to nothing, which is the shape a multinode upload produces. Run 2 + * exists to prove a purge never reaches past the runs it was given. + */ +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, html_url, created_at, date) + VALUES (1, 44000000001, 1, 'Run Sweep', null, now(), '2026-09-01'), + (2, 44000000002, 1, 'Run Sweep', null, now(), '2026-09-02'); + + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsv4', 'b200', 'sglang', 'fp8', 'none', false, 8, 8, 8, 8); + + INSERT INTO benchmark_results (id, workflow_run_id, config_id, date, isl, osl, conc, + benchmark_type, metrics) + VALUES (11, 1, 1, '2026-09-01', 8192, 1024, 32, 'single_turn', '{}'::jsonb), + (12, 1, 1, '2026-09-01', 8192, 1024, 64, 'single_turn', '{}'::jsonb), + (21, 2, 1, '2026-09-02', 8192, 1024, 32, 'single_turn', '{}'::jsonb); + + INSERT INTO gpu_metric_series (id, workflow_run_id, artifact_name, config_key, file_name, + vendor, csv_sha256, sample_count, gpu_count, started_at, ended_at) + VALUES (101, 1, 'gpu_metrics_a', 'a', 'a.csv', 'nvidia', 'sha-a', 1000, 8, + '2026-09-01T00:00:00Z', '2026-09-01T00:10:00Z'), + (102, 1, 'gpu_metrics_b', 'b', 'b.csv', 'nvidia', 'sha-b', 2000, 8, + '2026-09-01T00:00:00Z', '2026-09-01T00:20:00Z'), + (201, 2, 'gpu_metrics_c', 'c', 'c.csv', 'nvidia', 'sha-c', 4000, 8, + '2026-09-02T00:00:00Z', '2026-09-02T00:40:00Z'); + + INSERT INTO benchmark_result_gpu_metrics (benchmark_result_id, series_id) + VALUES (11, 101), (12, 101), (21, 201); + `); +}); + +describe('countRunTelemetry', () => { + it('sums the stored sample counts for the targeted runs only', async () => { + await expect(countRunTelemetry(sql, [1])).resolves.toEqual({ series: 2, samples: 3000 }); + await expect(countRunTelemetry(sql, [2])).resolves.toEqual({ series: 1, samples: 4000 }); + await expect(countRunTelemetry(sql, [1, 2])).resolves.toEqual({ series: 3, samples: 7000 }); + }); + + it('reports zero for a run with no telemetry and reads nothing for an empty list', async () => { + await db.exec('DELETE FROM gpu_metric_series WHERE workflow_run_id = 2'); + await expect(countRunTelemetry(sql, [2])).resolves.toEqual({ series: 0, samples: 0 }); + await expect(countRunTelemetry(sql, [])).resolves.toEqual({ series: 0, samples: 0 }); + }); + + it('does not delete anything', async () => { + await countRunTelemetry(sql, [1, 2]); + expect(await count('gpu_metric_series')).toBe(3); + }); +}); + +describe('deleteRunTelemetry', () => { + it('removes the run series and reports what went', async () => { + await expect(deleteRunTelemetry(sql, [1])).resolves.toEqual({ series: 2, samples: 3000 }); + expect(await count('gpu_metric_series')).toBe(1); + }); + + it('leaves other runs untouched', async () => { + await deleteRunTelemetry(sql, [1]); + await expect(countRunTelemetry(sql, [2])).resolves.toEqual({ series: 1, samples: 4000 }); + }); + + it('takes the point links with it, so no link outlives its series', async () => { + await deleteRunTelemetry(sql, [1]); + expect(await count('benchmark_result_gpu_metrics')).toBe(1); + }); + + it('is a no-op on an empty list', async () => { + await expect(deleteRunTelemetry(sql, [])).resolves.toEqual({ series: 0, samples: 0 }); + expect(await count('gpu_metric_series')).toBe(3); + }); +}); + +describe('unlinkPointTelemetry', () => { + it('drops only the links for the given points', async () => { + await expect(unlinkPointTelemetry(sql, [11])).resolves.toBe(1); + const rows = await db.query<{ benchmark_result_id: number }>( + 'SELECT benchmark_result_id FROM benchmark_result_gpu_metrics ORDER BY benchmark_result_id', + ); + expect(rows.rows.map((r) => r.benchmark_result_id)).toEqual([12, 21]); + }); + + it('keeps the series, which still belongs to a run that was not purged', async () => { + await unlinkPointTelemetry(sql, [11, 12]); + expect(await count('benchmark_result_gpu_metrics')).toBe(1); + await expect(countRunTelemetry(sql, [1])).resolves.toEqual({ series: 2, samples: 3000 }); + }); + + it('returns 0 for points that have no telemetry and for an empty list', async () => { + await expect(unlinkPointTelemetry(sql, [12, 999])).resolves.toBe(1); + await expect(unlinkPointTelemetry(sql, [999])).resolves.toBe(0); + await expect(unlinkPointTelemetry(sql, [])).resolves.toBe(0); + }); +}); + +describe('describeTelemetry', () => { + it('groups the sample count so a large loss reads as large', () => { + expect(describeTelemetry({ series: 23, samples: 207_345 })).toBe( + '23 gpu_metric_series (207,345 samples)', + ); + }); +}); diff --git a/packages/db/src/lib/telemetry-purge.ts b/packages/db/src/lib/telemetry-purge.ts new file mode 100644 index 000000000..13b2b76dd --- /dev/null +++ b/packages/db/src/lib/telemetry-purge.ts @@ -0,0 +1,100 @@ +/** + * Explicit deletion of PowerX telemetry during a purge. + * + * Migration 016 declares `on delete cascade` from `gpu_metric_series.workflow_run_id` + * to `workflow_runs` and from `benchmark_result_gpu_metrics.benchmark_result_id` to + * `benchmark_results`, so a purge already removed telemetry — silently, with no count + * in the preview and no line in the log. That is the wrong default for this data: + * past GitHub's 90-day artifact retention the stored samples are the only copy, which + * is the reason migration 016 exists at all. These helpers make the deletion explicit + * so the operator sees the cost before confirming and again in the transcript. + * + * The two levels differ on purpose: + * - A whole-run purge owns the series and deletes them (`deleteRunTelemetry`). + * - A point purge only drops the point→series links (`unlinkPointTelemetry`). The + * series stays attached to its workflow_run because `/api/gpu-metrics?runId=` + * reads series by run, not through the links, and other points of the same run + * may still reference it. + */ + +import type { Sql } from '../etl/db-utils.js'; + +export interface TelemetryPurgeCounts { + /** Rows in `gpu_metric_series`. */ + series: number; + /** Sum of `gpu_metric_series.sample_count` across those series. */ + samples: number; +} + +export const NO_TELEMETRY: TelemetryPurgeCounts = { series: 0, samples: 0 }; + +/** `count`/`sum` come back as strings over the wire; `sum` is null on an empty set. */ +function toCounts(row: { n: unknown; samples: unknown } | undefined): TelemetryPurgeCounts { + if (!row) return NO_TELEMETRY; + return { series: Number(row.n ?? 0), samples: Number(row.samples ?? 0) }; +} + +/** + * Telemetry a whole-run purge would destroy. Read-only, for the preview line. + * Sums the stored `sample_count` instead of counting `gpu_metric_samples` so the + * preview stays cheap against a table holding tens of millions of rows. + */ +export async function countRunTelemetry( + sql: Sql, + workflowRunIds: readonly number[], +): Promise { + if (workflowRunIds.length === 0) return NO_TELEMETRY; + const [row] = await sql` + SELECT count(*)::int AS n, coalesce(sum(sample_count), 0)::bigint AS samples + FROM gpu_metric_series + WHERE workflow_run_id = ANY(${[...workflowRunIds]}) + `; + return toCounts(row as { n: unknown; samples: unknown } | undefined); +} + +/** + * Delete the telemetry owned by these workflow_runs and report what went, so the + * caller can log it. Samples and per-GPU stats follow by cascade from the series. + * Call this before deleting the `workflow_runs` rows themselves. + */ +export async function deleteRunTelemetry( + sql: Sql, + workflowRunIds: readonly number[], +): Promise { + if (workflowRunIds.length === 0) return NO_TELEMETRY; + const [row] = await sql` + WITH deleted AS ( + DELETE FROM gpu_metric_series + WHERE workflow_run_id = ANY(${[...workflowRunIds]}) + RETURNING sample_count + ) + SELECT count(*)::int AS n, coalesce(sum(sample_count), 0)::bigint AS samples FROM deleted + `; + return toCounts(row as { n: unknown; samples: unknown } | undefined); +} + +/** + * Drop the point→series links for purged benchmark points and report how many went. + * Deliberately leaves `gpu_metric_series` in place: it belongs to the workflow_run, + * which is not being purged here. + */ +export async function unlinkPointTelemetry( + sql: Sql, + benchmarkResultIds: readonly number[], +): Promise { + if (benchmarkResultIds.length === 0) return 0; + const [row] = await sql` + WITH deleted AS ( + DELETE FROM benchmark_result_gpu_metrics + WHERE benchmark_result_id = ANY(${[...benchmarkResultIds]}) + RETURNING series_id + ) + SELECT count(*)::int AS n FROM deleted + `; + return Number((row as { n: unknown } | undefined)?.n ?? 0); +} + +/** One-line summary for preview and transcript output. */ +export function describeTelemetry(counts: TelemetryPurgeCounts): string { + return `${counts.series} gpu_metric_series (${counts.samples.toLocaleString('en-US')} samples)`; +} From c763540f1ceede432babe9c18f746cb766f810e7 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sun, 20 Sep 2026 21:04:51 -0700 Subject: [PATCH 016/103] fix: migrate before verify in apply-run-overrides and keep cache steps MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit This workflow fires on a push to run-overrides.ts, which can land before the next ingest dispatch has applied a pending migration. It ran admin:db:verify with no migrate step of its own, and verify-db counts every table it knows about, so it would fail on a schema the checked-out ref expects but production does not have yet. Added the same Run migrations step ingest-results.yml uses; migrations are idempotent by filename. The worse half was what a red verify did to the two steps after it. Neither carried if: always(), so a verify failure skipped both the production cache invalidation and the warmup while the overrides were already committed to the database. The dashboard then served pre-override data with a red workflow as the only signal. Both now run whenever the overrides step itself succeeded. 中文:该工作流在 run-overrides.ts 的 push 上触发,而这次 push 可能早于下一次 ingest 派发应用待处理的迁移。它自己没有 migrate 步骤就直接跑 admin:db:verify, 而 verify-db 会统计它已知的每一张表,于是会在一个"检出的 ref 期望、但生产还没有" 的 schema 上失败。现已加入与 ingest-results.yml 相同的 Run migrations 步骤;迁移 按文件名幂等。 更严重的一半是 verify 变红对其后两个步骤的影响。两者都没有 if: always(),所以 verify 一失败就会跳过生产缓存失效和预热,而此时 override 已经写进数据库了。仪表板 于是继续提供改写前的数据,唯一的信号只有一个变红的工作流。现在只要 override 步骤 本身成功,这两步就会执行。 --- .github/workflows/apply-run-overrides.yml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/.github/workflows/apply-run-overrides.yml b/.github/workflows/apply-run-overrides.yml index 749880e80..819f09aca 100644 --- a/.github/workflows/apply-run-overrides.yml +++ b/.github/workflows/apply-run-overrides.yml @@ -53,7 +53,18 @@ jobs: env: CYPRESS_INSTALL_BINARY: '0' + # This workflow fires on a push to run-overrides.ts, which can land before the + # next ingest dispatch has applied a pending migration. admin:db:verify counts + # every table it knows about, so without this step it fails on a schema the + # checked-out ref expects but production does not have yet. Same command and + # ordering as ingest-results.yml; migrations are idempotent by filename. + - name: Run migrations + env: + DATABASE_WRITE_URL: ${{ secrets.DATABASE_WRITE_URL }} + run: bun run admin:db:migrate --yes + - name: Apply run overrides + id: apply env: DATABASE_WRITE_URL: ${{ secrets.DATABASE_WRITE_URL }} RUN_ID: ${{ inputs.run_id }} @@ -69,10 +80,15 @@ jobs: DATABASE_WRITE_URL: ${{ secrets.DATABASE_WRITE_URL }} run: bun run admin:db:verify + # Once the overrides are in the database the CDN is stale, so these must run + # even if the verify step above fails. Skipping them leaves the dashboard + # serving pre-override data with only a red workflow as the signal. - name: Invalidate production cache + if: ${{ always() && steps.apply.outcome == 'success' }} env: INVALIDATE_SECRET: ${{ secrets.VERCEL_INVALIDATE_SECRET }} run: bun run admin:cache:invalidate https://inferencex.semianalysis.com - name: Warm production cache + if: ${{ always() && steps.apply.outcome == 'success' }} run: bun run admin:cache:warmup https://inferencex.semianalysis.com From 1e5d227acde10b20c346f79b9124bdc6bf62ff48 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Sun, 20 Sep 2026 21:05:11 -0700 Subject: [PATCH 017/103] fix: keep gpu_metrics digest errors from failing the whole ingest MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A gpu_metrics digest failure was recorded through recordDbError, which feeds tracker.skips.dbError, which the ingest writes into the publication manifest as an ingestError, which verify-power-publication folds into errors and exits non-zero on. That script is the required "Verify PowerX source, database and public API" step with no continue-on-error, so one malformed telemetry CSV would turn a whole production ingest red even though every benchmark row landed correctly. This failure mode does not exist before migration 016. Telemetry failures now have their own counter and their own recordTelemetryError, with its own print budget so telemetry noise cannot suppress real DB errors. The manifest reports them under telemetryWarnings, and the new fatalPublicationErrors makes the fatal set explicit at the one place that decides the exit code. What is lost on a digest failure is one point's PowerX tab, and admin:db:backfill-gpu-metrics --run can re-digest the artifact. Coverage: the two counters are proven independent, and fatalPublicationErrors is proven to ignore telemetry warnings while still failing on real ingest errors and verification mismatches. The catch site itself has no direct test — it sits inside the artifact loop of a script the suite only runs as a subprocess against a purged run, which returns before reaching it. 中文:gpu_metrics 摘要失败此前经 recordDbError 记录,进入 tracker.skips.dbError, 再被 ingest 作为 ingestError 写入发布 manifest,verify-power-publication 把它折进 errors 并以非零码退出。该脚本是必需的 "Verify PowerX source, database and public API" 步骤且没有 continue-on-error,因此一个格式错误的遥测 CSV 就能让整条生产 ingest 变红,哪怕每一行 benchmark 数据都正确落库。这个失败模式在迁移 016 之前 并不存在。 遥测失败现在有独立计数器和独立的 recordTelemetryError,并有自己的打印额度,避免 遥测噪声淹没真正的 DB 错误。manifest 将其归入 telemetryWarnings,新增的 fatalPublicationErrors 在决定退出码的唯一位置显式界定致命集合。摘要失败损失的只是 某一个点的 PowerX 标签页,用 admin:db:backfill-gpu-metrics --run 可以重新摘要 该产物。 覆盖范围:已验证两个计数器互相独立,并验证 fatalPublicationErrors 忽略遥测警告、 同时仍然对真正的 ingest 错误和校验不一致判为失败。catch 处本身没有直接测试——它 位于一个脚本的产物循环内部,而测试套件只以子进程方式对一个已 purge 的 run 运行该 脚本,那条路径在到达此处之前就返回了。 --- packages/db/src/etl/ingest-summary.ts | 1 + packages/db/src/etl/power-publication.test.ts | 28 ++++++++++++++++ packages/db/src/etl/power-publication.ts | 21 ++++++++++++ packages/db/src/etl/skip-tracker.test.ts | 21 ++++++++++++ packages/db/src/etl/skip-tracker.ts | 33 +++++++++++++++++++ packages/db/src/ingest-ci-run.test.ts | 1 + packages/db/src/ingest-ci-run.ts | 11 ++++++- packages/db/src/ingest-gcs-backup.ts | 10 ++++-- packages/db/src/verify-power-publication.ts | 12 ++++--- 9 files changed, 130 insertions(+), 8 deletions(-) diff --git a/packages/db/src/etl/ingest-summary.ts b/packages/db/src/etl/ingest-summary.ts index 16806cc47..684d7a1f5 100644 --- a/packages/db/src/etl/ingest-summary.ts +++ b/packages/db/src/etl/ingest-summary.ts @@ -38,6 +38,7 @@ export function printIngestSummaryFooter( ['unmapped hw', skips.unmappedHw], ['bad/empty zip', skips.badZip], ['DB errors', skips.dbError], + ['telemetry errors (non-fatal)', skips.telemetryError], ); const nonzeroSkipLines = skipLines.filter(([, count]) => count > 0); diff --git a/packages/db/src/etl/power-publication.test.ts b/packages/db/src/etl/power-publication.test.ts index b31ce0f98..60a95e3e1 100644 --- a/packages/db/src/etl/power-publication.test.ts +++ b/packages/db/src/etl/power-publication.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from 'vitest'; import { mapBenchmarkRow } from './benchmark-mapper'; import { createSkipTracker } from './skip-tracker'; import { + fatalPublicationErrors, powerPublicationPoint, verifyPowerPublication, type PublishedPowerRow, @@ -140,3 +141,30 @@ describe('PowerX publication', () => { expect(tracker.skips.failedRun).toBe(1); }); }); + +/** + * Regression for the ingest-reddening chain: a gpu_metrics digest failure used to + * be recorded as a DB error, reach the publication manifest's `ingestErrors`, and + * set exitCode 1 on the required "Verify PowerX source, database and public API" + * step — failing a production ingest whose benchmark data had landed fine. + */ +describe('fatalPublicationErrors', () => { + it('ignores telemetry warnings, so a bad gpu_metrics CSV cannot fail the ingest', () => { + expect( + fatalPublicationErrors({ telemetryWarnings: ['3 gpu_metrics digest errors'] }, []), + ).toEqual([]); + }); + + it('still fails on real ingest errors and on verification mismatches', () => { + expect( + fatalPublicationErrors( + { ingestErrors: ['2 database ingest errors'], telemetryWarnings: ['1 gpu_metrics'] }, + ['point 441871 expected absent, got 642'], + ), + ).toEqual(['2 database ingest errors', 'point 441871 expected absent, got 642']); + }); + + it('treats both fields as optional', () => { + expect(fatalPublicationErrors({}, [])).toEqual([]); + }); +}); diff --git a/packages/db/src/etl/power-publication.ts b/packages/db/src/etl/power-publication.ts index 5009468be..b0f359df8 100644 --- a/packages/db/src/etl/power-publication.ts +++ b/packages/db/src/etl/power-publication.ts @@ -46,8 +46,29 @@ export interface PowerPublicationManifest { runId: number; runAttempt: number; points: PowerPublicationPoint[]; + /** Fatal: verify-power-publication exits non-zero when this is non-empty. */ ingestErrors?: string[]; + /** + * Non-fatal: PowerX telemetry digest failures. Surfaced in the verification + * receipt so they stay visible, but they never fail the ingest — the benchmark + * rows landed, and the artifact can be re-digested by the backfill. + */ + telemetryWarnings?: string[]; } +/** + * The errors that fail an ingest. `telemetryWarnings` is deliberately not among + * them: a gpu_metrics digest failure costs one point's PowerX tab, while the + * benchmark rows it accompanies are already committed and the artifact can be + * re-digested by `admin:db:backfill-gpu-metrics`. Folding it in would let one + * malformed CSV turn a whole production ingest red. + */ +export function fatalPublicationErrors( + manifest: Pick, + verificationErrors: readonly string[], +): string[] { + return [...(manifest.ingestErrors ?? []), ...verificationErrors]; +} + export interface PublishedPowerRow extends Record { metrics: Record; } diff --git a/packages/db/src/etl/skip-tracker.test.ts b/packages/db/src/etl/skip-tracker.test.ts index e407db3a3..0e3ba64ee 100644 --- a/packages/db/src/etl/skip-tracker.test.ts +++ b/packages/db/src/etl/skip-tracker.test.ts @@ -9,6 +9,7 @@ describe('createSkipTracker', () => { expect(tracker.skips.unmappedHw).toBe(0); expect(tracker.skips.noIslOsl).toBe(0); expect(tracker.skips.dbError).toBe(0); + expect(tracker.skips.telemetryError).toBe(0); expect(tracker.skips.traceReplayMissing).toBe(0); }); @@ -53,6 +54,26 @@ describe('recordDbError', () => { }); }); +describe('recordTelemetryError', () => { + it('counts separately from dbError, which is the counter that fails the ingest', () => { + const tracker = createSkipTracker(); + tracker.recordTelemetryError('gpu_metrics for dsv4-b200', new Error('malformed CSV')); + tracker.recordTelemetryError('gpu_metrics for dsv4-b300', new Error('malformed CSV')); + expect(tracker.skips.telemetryError).toBe(2); + expect(tracker.skips.dbError).toBe(0); + }); + + it('keeps its own print budget, so telemetry noise cannot silence DB errors', () => { + const tracker = createSkipTracker(); + for (let i = 0; i < 15; i++) { + tracker.recordTelemetryError(`context ${i}`, new Error(`error ${i}`)); + } + tracker.recordDbError('availability', new Error('boom')); + expect(tracker.skips.telemetryError).toBe(15); + expect(tracker.skips.dbError).toBe(1); + }); +}); + describe('snapshot', () => { it('captures current counters', () => { const tracker = createSkipTracker(); diff --git a/packages/db/src/etl/skip-tracker.ts b/packages/db/src/etl/skip-tracker.ts index 5d485bf22..055a92b46 100644 --- a/packages/db/src/etl/skip-tracker.ts +++ b/packages/db/src/etl/skip-tracker.ts @@ -10,6 +10,15 @@ export interface Skips { noIslOsl: number; failedRun: number; dbError: number; + /** + * PowerX telemetry digest failures, counted apart from `dbError` because they + * are not fatal. The benchmark rows land either way; only the per-point + * telemetry tab is affected, and the artifact can be re-digested later by + * `admin:db:backfill-gpu-metrics`. Folding these into `dbError` would let one + * malformed `gpu_metrics_*` CSV turn the whole production ingest red, through + * the publication manifest that verify-power-publication treats as fatal. + */ + telemetryError: number; /** Agentic point whose sibling `agentic_` artifact had no trace_replay files. */ traceReplayMissing: number; } @@ -35,6 +44,15 @@ export interface SkipTracker { * @param err - The caught error. */ recordDbError: (context: string, err: Error) => void; + /** + * Record a non-fatal PowerX telemetry digest failure. Same printing and + * suppression as `recordDbError`, but increments `skips.telemetryError` so the + * ingest is not failed by it. + * + * @param context - Human-readable label for where the error occurred. + * @param err - The caught error. + */ + recordTelemetryError: (context: string, err: Error) => void; /** * Capture a point-in-time snapshot of the current skip counters and * unmapped-name sets. Used together with `diff()` to report per-artifact drops. @@ -76,12 +94,14 @@ export function createSkipTracker(): SkipTracker { noIslOsl: 0, failedRun: 0, dbError: 0, + telemetryError: 0, traceReplayMissing: 0, }; const unmappedModels = new Set(); const unmappedHws = new Set(); const unmappedPrecisions = new Set(); let dbErrorsPrinted = 0; + let telemetryErrorsPrinted = 0; return { skips, @@ -100,6 +120,19 @@ export function createSkipTracker(): SkipTracker { } }, + recordTelemetryError(context: string, err: Error): void { + skips.telemetryError++; + if (telemetryErrorsPrinted < MAX_DB_ERRORS) { + console.error(` [TELEMETRY] ${context}: ${err.message}`); + telemetryErrorsPrinted++; + if (telemetryErrorsPrinted === MAX_DB_ERRORS) { + console.error( + ' [TELEMETRY] further telemetry errors suppressed; count included in summary', + ); + } + } + }, + snapshot(): SkipSnapshot { return { model: skips.unmappedModel, diff --git a/packages/db/src/ingest-ci-run.test.ts b/packages/db/src/ingest-ci-run.test.ts index d3ad330b4..19812f372 100644 --- a/packages/db/src/ingest-ci-run.test.ts +++ b/packages/db/src/ingest-ci-run.test.ts @@ -53,6 +53,7 @@ describe('purged CI ingestion', () => { runAttempt, points: [], ingestErrors: [], + telemetryWarnings: [], }); } finally { fs.rmSync(dir, { recursive: true, force: true }); diff --git a/packages/db/src/ingest-ci-run.ts b/packages/db/src/ingest-ci-run.ts index debb9f463..c2ae094a7 100644 --- a/packages/db/src/ingest-ci-run.ts +++ b/packages/db/src/ingest-ci-run.ts @@ -715,7 +715,12 @@ async function main(): Promise { `${ingested.seriesSkipped} unchanged (${elapsed(gpuMetricsStart)})`, ); } catch (error: any) { - tracker.recordDbError(`gpu_metrics for ${configKey}`, error); + // Non-fatal on purpose: this point's benchmark rows are already + // committed and only its telemetry tab is affected, and + // `admin:db:backfill-gpu-metrics --run ` can re-digest the + // artifact later. Recording it as a DB error instead would reach + // the publication manifest and fail the whole production ingest. + tracker.recordTelemetryError(`gpu_metrics for ${configKey}`, error); } } } @@ -1110,6 +1115,10 @@ main() ...powerPublicationErrors, ...(tracker.skips.dbError ? [`${tracker.skips.dbError} database ingest errors`] : []), ], + // Reported but not fatal — see Skips.telemetryError. + telemetryWarnings: tracker.skips.telemetryError + ? [`${tracker.skips.telemetryError} gpu_metrics digest errors`] + : [], }, null, 2, diff --git a/packages/db/src/ingest-gcs-backup.ts b/packages/db/src/ingest-gcs-backup.ts index 709fdc9c3..b9cb76b59 100644 --- a/packages/db/src/ingest-gcs-backup.ts +++ b/packages/db/src/ingest-gcs-backup.ts @@ -104,8 +104,12 @@ interface WorkflowMapResult { changelogs: { baseRef: string; headRef: string; entries: ChangelogEntry[] }[]; /** True when the changelog declares evals-only — benchmark/stats data is dropped. */ evalsOnly: boolean; - /** Skip counts from mapping phase (dbError is tracked separately in phase 2). */ - localSkips: Omit; + /** + * Skip counts from the mapping phase. `dbError` is tracked separately in phase 2, + * and `telemetryError` never applies here: the GCS backup path ingests no + * `gpu_metrics_*` artifacts. + */ + localSkips: Omit; localUnmappedModels: Set; localUnmappedHws: Set; /** Pre-formatted [WARN] lines to print at the start of phase 2 for this dir. */ @@ -121,7 +125,7 @@ interface WriteResult { evalSamples: number; changelogs: number; warnings: string[]; - localSkips: Omit; + localSkips: Omit; localUnmappedModels: string[]; localUnmappedHws: string[]; } diff --git a/packages/db/src/verify-power-publication.ts b/packages/db/src/verify-power-publication.ts index fbd34ab68..1beb41e96 100644 --- a/packages/db/src/verify-power-publication.ts +++ b/packages/db/src/verify-power-publication.ts @@ -2,6 +2,7 @@ import fs from 'node:fs'; import { DB_MODEL_TO_DISPLAY } from '@semianalysisai/inferencex-constants'; import { createAdminSql } from './etl/db-utils'; import { + fatalPublicationErrors, verifyPowerPublication, type PowerPublicationManifest, type PublishedPowerRow, @@ -35,10 +36,10 @@ try { and (br.benchmark_type = 'agentic_traces' or (br.benchmark_type = 'single_turn' and br.isl in (1024, 8192) and br.osl = 1024)) `; - const errors = [ - ...(manifest.ingestErrors ?? []), - ...verifyPowerPublication(manifest.points, rows as unknown as PublishedPowerRow[], 'database'), - ]; + const errors = fatalPublicationErrors( + manifest, + verifyPowerPublication(manifest.points, rows as unknown as PublishedPowerRow[], 'database'), + ); const publicRows: PublishedPowerRow[] = []; const models = [ ...new Set(manifest.points.map((point) => DB_MODEL_TO_DISPLAY[String(point.identity.model)])), @@ -86,6 +87,9 @@ try { status: errors.length > 0 ? 'failed' : manifest.points.length > 0 ? 'matched' : 'no_power_points', errors, + // Kept out of `errors` on purpose: a telemetry digest failure costs one + // point's PowerX tab, not its benchmark data, so it must not fail the ingest. + telemetryWarnings: manifest.telemetryWarnings ?? [], }; fs.writeFileSync(`${manifestPath}.verification.json`, `${JSON.stringify(receipt, null, 2)}\n`); console.log(JSON.stringify(receipt, null, 2)); From af1a56f5c0d541526145bd22a1e55a61bdcd90db Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Mon, 21 Sep 2026 00:07:16 -0700 Subject: [PATCH 018/103] fix: keep a line label on every series instead of dropping crowded ones MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The anchor pass in placeLineLabels tried four fractions along each line and, if all of them collided, emitted visible:false. In the PowerX article's Fig 6 — roles mode with two ?unofficialrun= overlays, so nine lines whose anchors compete in one narrow band — that silently cost the GB300 overall series its pill. Its curve was still drawn, next to two labelled siblings of the same colour, so the reader could only identify it by elimination. The drop was a layering mistake. It decided visibility from a crude nominal box of collisionWidth/2 by 21px, while layoutPills runs right after with the real measured boxes, mirrors, shifts rows and clamps into the plot, and says in its own doc comment that an overlapped label beats a missing one. A rendered pill in this figure measures 223-330px against that 120px model, so the pass that gave up was the one least able to judge. The same function's pinned-anchor branch already emits visible:true unconditionally. A series with no clear slot is now deferred and placed after the clean ones, on whichever of its slots carries the least overlap, and it no longer anchors on points[0] — the axis-hugging index lineCandidates deliberately skips. Deferring matters on its own: a doomed series used to be able to reserve a slot that a later series could have had to itself. keepVisibleOnCollision is gone; it was the single-point special case of the rule this generalises. This ends the promise that line labels never overlap, stated in #132 and restated in #434. It was worth less than it cost: a dropped pill is silent, and with line labels on the PNG export omits the legend, so the series loses its only identifier. Verified at /inference?i_metric=y_measuredAvgPower&i_pcompare=roles&unofficialruns= 35319969159,35319956855 against the branch DB: 17 labels, all rendered, none overlapping, including the pill the figure was missing. The 12 pills still hidden there are the separate GH #470 de-duplication path, which is untouched. Gates: lint, fmt, typecheck, typography, 6,949 unit tests, 521 Cypress component tests. Three of the five new unit tests and the new component test fail on the old code. 中文:placeLineLabels 的锚点阶段沿每条线尝试四个位置,若全部碰撞就发出 visible:false。在 PowerX 文章的 Fig 6 中(roles 模式加两个 ?unofficialrun= 叠加, 九条线的锚点挤在同一窄带里),这让 GB300 整体序列悄悄丢掉了标签。它的曲线仍然 画着,旁边是两条同色且有标签的兄弟线,读者只能靠排除法辨认。 这个丢弃是分层错误。它用 collisionWidth/2 乘 21px 的粗略估算盒来决定可见性,而 紧随其后的 layoutPills 拿着真实测量盒做镜像、错行和边界钳制,其文档注释明确写着 重叠的标签也好过消失的标签。该图里一个实际渲染的标签宽 223 到 330px,而模型只按 120px 估算,所以放弃的恰恰是最没有判断力的那一遍。同一函数的固定锚点分支本来就 无条件发出 visible:true。 现在没有空闲槽位的序列会被推迟到干净标签之后放置,落在重叠代价最小的槽位上,并且 不再锚定到 points[0]——lineCandidates 刻意跳过的贴轴位置。推迟本身也有意义:原先 一个注定重叠的序列可能占掉后面序列本可独享的槽位。keepVisibleOnCollision 已删除, 它只是本规则的单点特例。 这终结了「折线标签永不重叠」的承诺,该承诺由 #132 提出、#434 重申。它的价值抵不上 代价:标签被丢弃是无声的,而开启折线标签后 PNG 导出会省略图例,序列就失去了唯一的 标识。 验证:在指向分支数据库的 /inference?i_metric=y_measuredAvgPower&i_pcompare=roles &unofficialruns=35319969159,35319956855 上,17 个标签全部渲染、互不覆盖,包括该图 原本缺失的那一个。页面上仍隐藏的 12 个标签来自独立的 GH #470 去重路径,未受影响。 闸门:lint、fmt、typecheck、typography、6,949 个单元测试、521 个 Cypress 组件测试。 五个新单元测试中的三个以及新增的组件测试在旧代码上失败。 --- .../cypress/component/power-compare.cy.tsx | 121 +++++++++++++++++- .../components/inference/ui/ScatterGraph.tsx | 2 - .../inference/ui/line-label-layer.test.ts | 118 ++++++++++++++++- .../inference/ui/line-label-layer.ts | 113 +++++++++++----- 4 files changed, 312 insertions(+), 42 deletions(-) diff --git a/packages/app/cypress/component/power-compare.cy.tsx b/packages/app/cypress/component/power-compare.cy.tsx index d14304e38..cca91215d 100644 --- a/packages/app/cypress/component/power-compare.cy.tsx +++ b/packages/app/cypress/component/power-compare.cy.tsx @@ -126,6 +126,8 @@ function mountCompare( metric?: CompareMetric; lineLabels?: boolean; precisions?: Precision[]; + /** Extra `?unofficialrun=` runs beyond the default one, for multi-run tests. */ + extraRuns?: { id: number; url: string; hwKeys: string[] }[]; } = {}, ) { const metricKey = options.metric ?? 'y_measuredAvgPower'; @@ -165,9 +167,24 @@ function mountCompare( unofficial: options.overlay ? { isUnofficialRun: true, - activeOverlayHwTypes: new Set(['h100']), - allOverlayHwTypes: new Set(['h100']), - runIndexByUrl: { [OVERLAY_RUN_URL]: 0, [String(OVERLAY_RUN_ID)]: 0 }, + activeOverlayHwTypes: new Set([ + 'h100', + ...(options.extraRuns ?? []).flatMap((run) => run.hwKeys), + ]), + allOverlayHwTypes: new Set([ + 'h100', + ...(options.extraRuns ?? []).flatMap((run) => run.hwKeys), + ]), + runIndexByUrl: { + [OVERLAY_RUN_URL]: 0, + [String(OVERLAY_RUN_ID)]: 0, + ...Object.fromEntries( + (options.extraRuns ?? []).flatMap((run, index) => [ + [run.url, index + 1], + [String(run.id), index + 1], + ]), + ), + }, unofficialRunInfos: [ { id: OVERLAY_RUN_ID, @@ -180,6 +197,17 @@ function mountCompare( status: 'completed', isNonMainBranch: true, }, + ...(options.extraRuns ?? []).map((run) => ({ + id: run.id, + name: `powerx-compare-${run.id}`, + branch: `powerx-compare-${run.id}`, + sha: 'abc001', + createdAt: '2026-09-02T00:00:00Z', + url: run.url, + conclusion: 'success' as const, + status: 'completed' as const, + isNonMainBranch: true, + })), ], } : {}, @@ -500,6 +528,93 @@ describe('ScatterGraph power comparison series', () => { }); }); + /** + * Regression for the PowerX article's Fig 6. Two `?unofficialrun=` overlays in + * roles mode draw six lines, and the anchor pass used to hide whichever series + * ran out of anchor slots — in the exported figure that was the GB300 overall + * line, leaving a drawn curve the reader could only identify by elimination. + * + * `crowdedCurve` reproduces the shape that causes it, which a gently spread + * curve does not: power falls steeply with interactivity, so every series' + * anchors land in the top slice of a tall y domain and compete for the same + * few x positions. Pills stay in the DOM at opacity 0 when hidden, so counting + * nodes cannot see this bug; the assertions below are on `data-visible`. + */ + const crowdedCurve = (hwKey: string, runUrl: string, offset: number): InferenceData[] => + [ + [8, 840, 848, 832], + [16, 700, 712, 690], + [32, 470, 486, 458], + [64, 330, 344, 318], + [128, 250, 262, 240], + ].map(([x, measured, prefill, decode]) => + createMockInferenceData({ + hwKey, + x, + conc: x, + y: measured + offset, + precision: Precision.FP4, + run_url: runUrl, + disagg: true, + measuredAvgPower: metric(measured + offset), + measuredPrefillAvgPower: metric(prefill + offset), + measuredDecodeAvgPower: metric(decode + offset), + gpuProvisionedWatts: metric(1000), + utilityProvisionedWatts: metric(1710), + }), + ); + + it('keeps a visible pill on all three role series of every ?unofficialrun= overlay', () => { + const secondRunId = 27182818284; + const secondRunUrl = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${secondRunId}`; + const overlay = [ + ...expandPowerCompareSeries( + crowdedCurve('h100', OVERLAY_RUN_URL, 0), + 'y_measuredAvgPower', + 'roles', + ), + ...expandPowerCompareSeries( + crowdedCurve('b200', secondRunUrl, 6), + 'y_measuredAvgPower', + 'roles', + ), + ]; + mountCompare([], { + overlay, + lineLabels: true, + extraRuns: [{ id: secondRunId, url: secondRunUrl, hwKeys: ['b200'] }], + }); + + cy.get(lineLabels).should('have.length', 6); + cy.get(`${lineLabels}[data-visible="1"]`).should('have.length', 6); + // The two "overall" pills are the ones the old anchor pass dropped. + cy.get(`${svg} ${labelSelector('')}[data-visible="1"]`).should('have.length', 2); + // A kept pill is only worth keeping if it is on the chart. (layoutPills' + // clamp is unit-tested against the exact clip rect in + // line-label-pill-bounds.test.ts; this is the coarse check that nothing was + // pushed off the chart entirely.) + cy.get(svg).then(($svg) => { + const chart = $svg[0].getBoundingClientRect(); + cy.get(`${lineLabels}[data-visible="1"] .ll-bg`).each(($bg) => { + const box = $bg[0].getBoundingClientRect(); + expect(box.left, 'pill left inside chart').to.be.at.least(chart.left - 1); + expect(box.right, 'pill right inside chart').to.be.at.most(chart.right + 1); + expect(box.top, 'pill top inside chart').to.be.at.least(chart.top - 1); + expect(box.bottom, 'pill bottom inside chart').to.be.at.most(chart.bottom + 1); + }); + }); + // Both runs contribute all three roles, keyed by line so the runs stay apart. + cy.get(`${lineLabels}[data-visible="1"]`).then(($labels) => { + const byLine = [...$labels].map((node) => ({ + line: (node as HTMLElement).dataset.lineKey ?? '', + text: pillText(node), + })); + expect(new Set(byLine.map((entry) => entry.line)).size, 'distinct line keys').to.eq(6); + expect(byLine.filter((entry) => /H100/u.test(entry.text)).length).to.eq(3); + expect(byLine.filter((entry) => /B200/u.test(entry.text)).length).to.eq(3); + }); + }); + it('labels ?unofficialrun= comparison siblings with the marked hardware and the variant', () => { const overlay = expandPowerCompareSeries( measuredCurve('h100', { run_url: OVERLAY_RUN_URL }), diff --git a/packages/app/src/components/inference/ui/ScatterGraph.tsx b/packages/app/src/components/inference/ui/ScatterGraph.tsx index 301e55652..e39bd894f 100644 --- a/packages/app/src/components/inference/ui/ScatterGraph.tsx +++ b/packages/app/src/components/inference/ui/ScatterGraph.tsx @@ -2911,7 +2911,6 @@ const ScatterGraph = React.memo( )}${suffix}`, color: ir.getCssColor(ir.resolveColor(entry.hw)), points: entry.points, - keepVisibleOnCollision: entry.points.length === 1, }; }); // Runs drawing the same hardware need a run tag on their pills. @@ -3221,7 +3220,6 @@ const ScatterGraph = React.memo( ...entry, label: '', color: '', - keepVisibleOnCollision: entry.points.length === 1, }), ); const overlaySeries: LineLabelSeries[] = Object.entries( diff --git a/packages/app/src/components/inference/ui/line-label-layer.test.ts b/packages/app/src/components/inference/ui/line-label-layer.test.ts index 33bb77256..0518b4ea5 100644 --- a/packages/app/src/components/inference/ui/line-label-layer.test.ts +++ b/packages/app/src/components/inference/ui/line-label-layer.test.ts @@ -14,17 +14,12 @@ interface Point { y: number; } -const series = ( - key: string, - points: Point[], - keepVisibleOnCollision = false, -): LineLabelSeries => ({ +const series = (key: string, points: Point[]): LineLabelSeries => ({ key, seriesId: key, label: key, color: '#000', points, - keepVisibleOnCollision, }); const identity = (value: number) => value; @@ -185,4 +180,115 @@ describe('line-label placement', () => { expect(labels[0]).toMatchObject({ x: 5, y: 7, visible: true }); }); + + /** + * Regression for the PowerX roles-mode figure: with two `?unofficialrun=` + * overlays the chart draws nine lines whose first points sit in one narrow + * band, every anchor slot is taken, and one series lost its pill entirely — + * leaving a drawn line the reader could not identify. + * + * `band` reproduces that shape: same x sweep for everyone, y values packed + * inside one collision height. + */ + describe('crowded series', () => { + const band = (key: string, y: number) => + series( + key, + [0, 25, 50, 75, 100].map((x) => ({ x, y })), + ); + const KEYS = ['a', 'b', 'c', 'd', 'e', 'f']; + const crowd = () => + placeLineLabels( + KEYS.map((key, index) => band(key, 100 + index * 4)), + identity, + identity, + { collisionWidth: 120 }, + ); + + it('keeps a pill for every series when every anchor slot is taken', () => { + const labels = crowd(); + + expect(labels).toHaveLength(KEYS.length); + expect(labels.every((label) => label.visible)).toBe(true); + expect(new Set(labels.map((label) => label.key))).toEqual(new Set(KEYS)); + }); + + it("never anchors a crowded label on the line's first point", () => { + // x = 0 is points[0], the axis-hugging index lineCandidates excludes. + expect(crowd().every((label) => label.x > 0)).toBe(true); + }); + + it('puts a crowded label on the least crowded of its slots', () => { + // Slots 100px apart so a 40px collision reach cannot bleed between them. + // Every slot is occupied, so there is no clear choice; slot 200 carries + // one obstacle against three on the first slot tried and two on the rest, + // so only a scored fallback lands there. + const spread = series( + 'only', + [0, 100, 200, 300, 400].map((x) => ({ x, y: 100 })), + ); + const labels = placeLineLabels([spread], identity, identity, { + collisionWidth: 40, + collisionHeight: 20, + obstacles: [ + { x: 100, y: 100, halfW: 20 }, + { x: 100, y: 105, halfW: 20 }, + { x: 100, y: 110, halfW: 20 }, + { x: 200, y: 100, halfW: 20 }, + { x: 300, y: 100, halfW: 20 }, + { x: 300, y: 105, halfW: 20 }, + { x: 400, y: 100, halfW: 20 }, + { x: 400, y: 105, halfW: 20 }, + ], + }); + + expect(labels[0]).toMatchObject({ x: 200, visible: true }); + }); + + /** + * `clean` starts below the whole band, so the y sort puts it LAST. Both + * tests below would pass trivially if it sorted first; placing it last is + * what makes them discriminate against the obvious wrong fix, which is to + * emit a crowded label in sort order the moment its slots run out. + */ + const clean = () => + series('clean', [ + { x: 0, y: 900 }, + { x: 100, y: 1000 }, + ]); + + it('lets a clean label keep its slot when earlier series have no slot at all', () => { + const alone = placeLineLabels([clean()], identity, identity, { collisionWidth: 120 }); + const withCrowd = placeLineLabels( + [...KEYS.map((key, index) => band(key, 100 + index * 4)), clean()], + identity, + identity, + { collisionWidth: 120 }, + ); + + // A crowded series must not consume a slot that a later series could + // have had to itself, so `clean` lands exactly where it would alone. + expect(withCrowd.find((label) => label.key === 'clean')).toMatchObject({ + x: alone[0].x, + y: alone[0].y, + visible: true, + }); + }); + + it('emits crowded labels after the clean ones regardless of sort order', () => { + // updateRenderedLineLabels lays pills out in this order, so the labels + // that will have to move are handed to it last. + const labels = placeLineLabels( + [...KEYS.map((key, index) => band(key, 100 + index * 4)), clean()], + identity, + identity, + { collisionWidth: 120 }, + ); + + const keys = labels.map((label) => label.key); + expect(keys[0]).toBe('a'); + expect(keys[1]).toBe('clean'); + expect(keys.slice(2)).toEqual(['b', 'c', 'd', 'e', 'f']); + }); + }); }); diff --git a/packages/app/src/components/inference/ui/line-label-layer.ts b/packages/app/src/components/inference/ui/line-label-layer.ts index 2db4806a8..7ac0d0d85 100644 --- a/packages/app/src/components/inference/ui/line-label-layer.ts +++ b/packages/app/src/components/inference/ui/line-label-layer.ts @@ -15,7 +15,6 @@ export interface LineLabelSeries { label: string; color: string; points: readonly TPoint[]; - keepVisibleOnCollision?: boolean; } export interface LineLabelPlacement { @@ -25,6 +24,12 @@ export interface LineLabelPlacement { color: string; x: number; y: number; + /** + * `placeLineLabels` always emits `true`: every series it is given keeps its + * pill, overlapping if it must. The only producer of `false` is the caller's + * de-duplication pass, which keeps a hidden data-join entry for a curve that + * lost the one-label-per-hardware contest (GH #470). + */ visible: boolean; } @@ -145,16 +150,16 @@ interface PillLayoutItem { * its anchor, and both mirrors together. Every candidate is clamped into * `bounds` before the overlap test, so nothing leaves the plot. When every * mirrored candidate collides, nearby rows are tried before the default spot - * is kept: an overlapped label is still - * better than a missing one, and the fallback matches what the anchor pass - * already tolerates for pinned anchors. + * is kept: an overlapped label is still better than a missing one, and the + * fallback matches what the anchor pass already tolerates. * * With no bounds — a chart that clips nothing — the anchor offset is applied * unchanged and the collision pass is skipped, preserving that chart's * existing layout. * * Hidden pills get their default transform and occupy no space, so a label - * that later becomes visible reappears where the anchor pass put it. + * that later becomes visible reappears where the anchor pass put it. Only the + * caller's de-duplication pass hides pills; the anchor pass never does. */ function layoutPills( items: readonly PillLayoutItem[], @@ -318,6 +323,21 @@ function lineCandidates( return candidates; } +/** + * Anchor one pill per series along its line. + * + * Each series tries `ANCHOR_SLOTS` fractions along its own points, rotated by + * its index so converging curves spread out instead of stacking at the + * endpoint. A series that finds a clear slot takes it. A series that finds none + * is deferred and placed afterwards on its least crowded slot, so it never + * steals a clear slot from a series that could have used it. + * + * Every series gets a visible pill. The overlap that survives here is resolved + * by `layoutPills`, which runs later with the pills' real measured boxes; the + * crude nominal box used here is far too small to decide that a label is + * unplaceable — a rendered pill is routinely two to three times + * `collisionWidth`. + */ export function placeLineLabels( series: readonly LineLabelSeries[], xScale: (value: number) => number, @@ -345,6 +365,35 @@ export function placeLineLabels( Math.abs(other.y - y) < collisionHeight && Math.abs(other.x - x) < other.halfW + labelHalfWidth, ); + /** + * Nominal overlap area against the labels already placed. The same crude box + * model as `collides`, scored instead of thresholded, so a slot that clips one + * neighbour is preferred over one that sits on three. + */ + const collisionCost = (x: number, y: number) => + placed.reduce((cost, other) => { + const dx = other.halfW + labelHalfWidth - Math.abs(other.x - x); + const dy = collisionHeight - Math.abs(other.y - y); + return dx > 0 && dy > 0 ? cost + dx * dy : cost; + }, 0); + + const emit = (entry: LineLabelSeries, point: TPoint) => { + const x = xScale(point.x); + const y = yScale(point.y); + placed.push({ x, y, halfW: labelHalfWidth }); + result.push({ + key: entry.key, + seriesId: entry.seriesId, + label: entry.label, + color: entry.color, + x, + y, + visible: true, + }); + }; + + /** Series with no clear slot, deferred to a second pass — see below. */ + const crowded: { entry: LineLabelSeries; candidates: TPoint[] }[] = []; for (const [seriesIndex, entry] of sorted.entries()) { if (entry.points.length === 0) continue; @@ -375,35 +424,37 @@ export function placeLineLabels( const candidate = candidates.find((point) => !collides(xScale(point.x), yScale(point.y))); if (candidate) { - const x = xScale(candidate.x); - const y = yScale(candidate.y); - placed.push({ x, y, halfW: labelHalfWidth }); - result.push({ - key: entry.key, - seriesId: entry.seriesId, - label: entry.label, - color: entry.color, - x, - y, - visible: true, - }); + emit(entry, candidate); continue; } - const fallback = entry.points[0]; - const x = xScale(fallback.x); - const y = yScale(fallback.y); - const visible = entry.keepVisibleOnCollision === true; - if (visible) placed.push({ x, y, halfW: labelHalfWidth }); - result.push({ - key: entry.key, - seriesId: entry.seriesId, - label: entry.label, - color: entry.color, - x, - y, - visible, - }); + // No clear slot. Defer rather than claim one now: a series that is going to + // overlap something must not take a slot a later series could have had to + // itself. + crowded.push({ entry, candidates }); + } + + // Every series keeps a pill. `layoutPills` runs after this with the real + // measured boxes and can still mirror it, shift it a row and clamp it into the + // plot — "an overlapped label is still better than a missing one". Emitting the + // crowded ones last also hands that pass the clean labels first, so the crowded + // ones do the moving. + // + // This is where the chart stopped promising that line labels never overlap + // (#132 introduced the drop as the only way to honour that, #434 restated it). + // The promise was worth less than it cost: a dropped pill is silent, and with + // line labels on, PNG export omits the legend, so the series loses its only + // identifier. An overlapping pill at least announces itself. + for (const { entry, candidates } of crowded) { + emit( + entry, + candidates.reduce((best, point) => + collisionCost(xScale(point.x), yScale(point.y)) < + collisionCost(xScale(best.x), yScale(best.y)) + ? point + : best, + ), + ); } return result; From 9bbc2290ca4af9a4d048aac8e934860086571e78 Mon Sep 17 00:00:00 2001 From: Wenyao Gao Date: Mon, 21 Sep 2026 00:28:54 -0700 Subject: [PATCH 019/103] refactor: remove unused PowerX telemetry helpers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Remove unused availability and sample-omission paths, identity wrappers, test-only label/source helpers, and write-only telemetry fields. Keep label formatting and nested-source coverage on the production helpers, preserve artifact failure isolation, and refresh the unchanged API contract digest. 中文:删除未使用的 PowerX 遥测查询模式、包装函数与只写不读的字段; 将标签格式及嵌套路径测试迁到实际生产入口,保留逐产物错误隔离, 并同步 API 路由摘要。此次仅落实精简清单 1–8。 --- docs/powerx-permanent-view.md | 2 +- packages/app/src/app/api/gpu-metrics/route.ts | 19 +++------ .../components/gpu-power/GpuPowerChart.tsx | 1 - .../agentic-point/power-telemetry-view.tsx | 2 - .../inference/utils/power-compare.test.ts | 36 ++++++++--------- .../inference/utils/power-compare.ts | 18 --------- .../inference/utils/powerTimeline.test.ts | 36 ++++++----------- .../inference/utils/powerTimeline.ts | 8 ---- packages/app/src/lib/api-route-catalog.ts | 2 +- packages/db/src/etl/gpu-metrics-ingest.ts | 32 +++++++-------- packages/db/src/queries/gpu-metrics.test.ts | 32 +-------------- packages/db/src/queries/gpu-metrics.ts | 40 +++++-------------- 12 files changed, 63 insertions(+), 165 deletions(-) diff --git a/docs/powerx-permanent-view.md b/docs/powerx-permanent-view.md index 9559fb31e..0325c4a84 100644 --- a/docs/powerx-permanent-view.md +++ b/docs/powerx-permanent-view.md @@ -146,7 +146,7 @@ was previously split on `_`. Siblings keep the hardware colour (overlay runs kee colour) and take a per-variant `stroke-dasharray` (`powerVariantDash`); clone points render at 0.6 opacity behind their base and carry `data-power-variant`. Line labels are placed per hardware _and_ sibling: the base series keeps its plain hardware label, while a sibling appends -` · ` (`powerLineLabel`: TDP, All-in, PUE modeled, Prefill GPUs, …) and, when a +` · ` (`powerLineLabelSuffix`: TDP, All-in, PUE modeled, Prefill GPUs, …) and, when a boundary is flat on a watts axis, its shared value (`B300 (SGLang) · TDP 1.2 kW`), so an exported PNG explains its dashed lines without the legend; each pill carries `data-series-id` (`::` for a sibling) and `data-power-variant`. The legend appends one diff --git a/packages/app/src/app/api/gpu-metrics/route.ts b/packages/app/src/app/api/gpu-metrics/route.ts index 87488e11c..ac13862cb 100644 --- a/packages/app/src/app/api/gpu-metrics/route.ts +++ b/packages/app/src/app/api/gpu-metrics/route.ts @@ -196,25 +196,18 @@ async function downloadBundle( */ async function runJob(job: TelemetryJob, githubToken: string): Promise { try { - return await runJobUnguarded(job, githubToken); + if (job.kind === 'csv') { + const parsed = await downloadArtifact(job.artifact, githubToken); + return parsed ? { kind: 'csv', parsed } : null; + } + const series = await downloadBundle(job.artifact, githubToken); + return series ? { kind: 'bundle', series } : null; } catch (error) { console.warn(`Failed to read artifact ${job.artifact.name}:`, error); return null; } } -async function runJobUnguarded( - job: TelemetryJob, - githubToken: string, -): Promise { - if (job.kind === 'csv') { - const parsed = await downloadArtifact(job.artifact, githubToken); - return parsed ? { kind: 'csv', parsed } : null; - } - const series = await downloadBundle(job.artifact, githubToken); - return series ? { kind: 'bundle', series } : null; -} - /** Downloads in listing order with a bounded number of requests in flight. */ async function downloadTelemetry( jobs: TelemetryJob[], diff --git a/packages/app/src/components/gpu-power/GpuPowerChart.tsx b/packages/app/src/components/gpu-power/GpuPowerChart.tsx index addb25953..a1209ec78 100644 --- a/packages/app/src/components/gpu-power/GpuPowerChart.tsx +++ b/packages/app/src/components/gpu-power/GpuPowerChart.tsx @@ -60,7 +60,6 @@ const STRINGS = { /** A second time series drawn over the telemetry on its own right-hand axis. */ export interface TelemetryOverlaySeries { - key: string; label: string; unit: string; color: string; diff --git a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx index a2c437f72..831fa1cc0 100644 --- a/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx +++ b/packages/app/src/components/inference/agentic-point/power-telemetry-view.tsx @@ -107,7 +107,6 @@ export function collectorLabel(series: Pick { expect(formatWatts(100000)).toBe('100 kW'); }); - it('leaves the base label alone and suffixes a sibling with its variant and flat watts', () => { - const label = 'B300 (SGLang)'; + it('omits the base suffix and labels a sibling with its variant and flat watts', () => { const tdp = basis('gpu-provisioned'); - expect(powerLineLabel(label, undefined, { isBase: true, locale: 'en' })).toBe(label); - expect(powerLineLabel(label, null, { isBase: false, locale: 'en' })).toBe(label); + expect(powerLineLabelSuffix(undefined, { isBase: true, locale: 'en' })).toBe(''); + expect(powerLineLabelSuffix(null, { isBase: false, locale: 'en' })).toBe(''); // A variant that is itself the base series keeps the plain label. - expect(powerLineLabel(label, tdp, { isBase: true, locale: 'en', flatWatts: 1200 })).toBe(label); - expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en' })).toBe('B300 (SGLang) · TDP'); - expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en', flatWatts: 700 })).toBe( - 'B300 (SGLang) · TDP 700 W', + expect(powerLineLabelSuffix(tdp, { isBase: true, locale: 'en', flatWatts: 1200 })).toBe(''); + expect(powerLineLabelSuffix(tdp, { isBase: false, locale: 'en' })).toBe(' · TDP'); + expect(powerLineLabelSuffix(tdp, { isBase: false, locale: 'en', flatWatts: 700 })).toBe( + ' · TDP 700 W', ); - expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en', flatWatts: 1370 })).toBe( - 'B300 (SGLang) · TDP 1.37 kW', + expect(powerLineLabelSuffix(tdp, { isBase: false, locale: 'en', flatWatts: 1370 })).toBe( + ' · TDP 1.37 kW', ); expect( - powerLineLabel(label, basis('utility-provisioned'), { + powerLineLabelSuffix(basis('utility-provisioned'), { isBase: false, locale: 'zh', flatWatts: 19200, }), - ).toBe('B300 (SGLang) · 全站 19.2 kW'); + ).toBe(' · 全站 19.2 kW'); // Non-finite or null watts drop the value, never print NaN. - expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en', flatWatts: null })).toBe( - 'B300 (SGLang) · TDP', + expect(powerLineLabelSuffix(tdp, { isBase: false, locale: 'en', flatWatts: null })).toBe( + ' · TDP', ); - expect(powerLineLabel(label, tdp, { isBase: false, locale: 'en', flatWatts: NaN })).toBe( - 'B300 (SGLang) · TDP', + expect(powerLineLabelSuffix(tdp, { isBase: false, locale: 'en', flatWatts: NaN })).toBe( + ' · TDP', ); - expect(powerLineLabel(label, role('decode'), { isBase: false, locale: 'en' })).toBe( - 'B300 (SGLang) · Decode GPUs', + expect(powerLineLabelSuffix(role('decode'), { isBase: false, locale: 'en' })).toBe( + ' · Decode GPUs', ); // The suffix alone is what the renderer splits into its own text segment. expect(powerLineLabelSuffix(role('prefill'), { isBase: false, locale: 'zh' })).toBe( diff --git a/packages/app/src/components/inference/utils/power-compare.ts b/packages/app/src/components/inference/utils/power-compare.ts index 19300b8cd..d0dea0b73 100644 --- a/packages/app/src/components/inference/utils/power-compare.ts +++ b/packages/app/src/components/inference/utils/power-compare.ts @@ -25,7 +25,6 @@ import { import { getMeasuredMetricConfig } from '../measured-metric-config'; import type { InferenceData, PowerCompare, PowerRole, PowerVariant } from '../types'; -import { reconstructedRoleEnergy } from './role-energy'; export const POWER_COMPARE_MODES = [ 'none', @@ -286,18 +285,6 @@ export function powerLineLabelSuffix( return `${LINE_LABEL_SUFFIX_SEPARATOR}${powerVariantShortLabel(variant, opts.locale)}${watts}`; } -/** - * Line-label text for one drawn series: the hardware label alone for the base - * series, `