diff --git a/.github/workflows/apply-run-overrides.yml b/.github/workflows/apply-run-overrides.yml index 749880e80..819f09aca 100644 --- a/.github/workflows/apply-run-overrides.yml +++ b/.github/workflows/apply-run-overrides.yml @@ -53,7 +53,18 @@ jobs: env: CYPRESS_INSTALL_BINARY: '0' + # This workflow fires on a push to run-overrides.ts, which can land before the + # next ingest dispatch has applied a pending migration. admin:db:verify counts + # every table it knows about, so without this step it fails on a schema the + # checked-out ref expects but production does not have yet. Same command and + # ordering as ingest-results.yml; migrations are idempotent by filename. + - name: Run migrations + env: + DATABASE_WRITE_URL: ${{ secrets.DATABASE_WRITE_URL }} + run: bun run admin:db:migrate --yes + - name: Apply run overrides + id: apply env: DATABASE_WRITE_URL: ${{ secrets.DATABASE_WRITE_URL }} RUN_ID: ${{ inputs.run_id }} @@ -69,10 +80,15 @@ jobs: DATABASE_WRITE_URL: ${{ secrets.DATABASE_WRITE_URL }} run: bun run admin:db:verify + # Once the overrides are in the database the CDN is stale, so these must run + # even if the verify step above fails. Skipping them leaves the dashboard + # serving pre-override data with only a red workflow as the signal. - name: Invalidate production cache + if: ${{ always() && steps.apply.outcome == 'success' }} env: INVALIDATE_SECRET: ${{ secrets.VERCEL_INVALIDATE_SECRET }} run: bun run admin:cache:invalidate https://inferencex.semianalysis.com - name: Warm production cache + if: ${{ always() && steps.apply.outcome == 'success' }} run: bun run admin:cache:warmup https://inferencex.semianalysis.com diff --git a/.github/workflows/ingest-agentic-results.yml b/.github/workflows/ingest-agentic-results.yml index b81cd68b0..3dcd23773 100644 --- a/.github/workflows/ingest-agentic-results.yml +++ b/.github/workflows/ingest-agentic-results.yml @@ -31,6 +31,11 @@ on: description: InferenceX Actions run ID to ingest required: true type: string + require-power: + description: Require the versioned PowerX manifest and all declared evidence + required: false + default: false + type: boolean run-attempt: description: InferenceX Actions run attempt to ingest required: false @@ -85,6 +90,11 @@ on: description: InferenceX Actions run ID to ingest required: true type: string + require-power: + description: Require the versioned PowerX manifest and all declared evidence + required: false + default: false + type: boolean run-attempt: description: InferenceX Actions run attempt to ingest required: false @@ -276,6 +286,7 @@ jobs: github.event.client_payload.run-id || inputs.run-id }} ARTIFACTS_PATH: ${{ github.workspace }}/artifacts INGEST_REPO: SemiAnalysisAI/InferenceX + INGEST_REQUIRE_POWER: ${{ github.event.client_payload.require-power == true || inputs.require-power == true }} run: bun run admin:db:prepare:ci - name: Run migrations @@ -288,6 +299,7 @@ jobs: INGEST_RUN_ATTEMPT: ${{ steps.artifacts.outputs.merge-run-attempt }} INGEST_ARTIFACTS_PATH: ${{ github.workspace }}/artifacts INGEST_REPO: SemiAnalysisAI/InferenceX + INGEST_REQUIRE_POWER: ${{ github.event.client_payload.require-power == true || inputs.require-power == true }} UNMAPPED_ENTITIES_OUTPUT: ${{ github.workspace }}/unmapped-entities.json POWER_PUBLICATION_MANIFEST: ${{ github.workspace }}/power-publication.json run: bun run admin:db:ingest:ci diff --git a/.github/workflows/ingest-results.yml b/.github/workflows/ingest-results.yml index 04b253e34..d66e50a8f 100644 --- a/.github/workflows/ingest-results.yml +++ b/.github/workflows/ingest-results.yml @@ -60,6 +60,7 @@ jobs: github.event.client_payload.run-id }} ARTIFACTS_PATH: ${{ github.workspace }}/artifacts INGEST_REPO: SemiAnalysisAI/InferenceX + INGEST_REQUIRE_POWER: ${{ github.event.client_payload.require-power == true }} run: bun run admin:db:prepare:ci - name: Run migrations @@ -75,6 +76,7 @@ jobs: INGEST_RUN_ATTEMPT: ${{ steps.artifacts.outputs.merge-run-attempt }} INGEST_ARTIFACTS_PATH: ${{ github.workspace }}/artifacts INGEST_REPO: SemiAnalysisAI/InferenceX + INGEST_REQUIRE_POWER: ${{ github.event.client_payload.require-power == true }} POWER_PUBLICATION_MANIFEST: ${{ github.workspace }}/power-publication.json UNMAPPED_ENTITIES_OUTPUT: ${{ github.workspace }}/unmapped-entities.json run: bun run admin:db:ingest:ci diff --git a/AGENTS.md b/AGENTS.md index baf40e2dc..61be9ec0e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -85,6 +85,7 @@ API routes (`packages/app/src/app/api/v1/`): - `reliability` — raw `ReliabilityRow[]` - `evaluations` — raw `EvalRow[]` - `server-log` — retrieve benchmark runtime logs +- `gpu-metrics-point?id=N` — PowerX chip telemetry (samples + per-GPU digest) linked to one benchmark point - `invalidate` — invalidate API cache (admin; `?scope=collectivex` purges only that scope) - `collectivex/latest`, `collectivex/runs`, `collectivex/runs/[runId]` — CollectiveX sweep data from a **separate** Neon DB, populated lazily on read from GitHub Actions artifacts and served diff --git a/bun.lock b/bun.lock index 29d749498..ea567a09d 100644 --- a/bun.lock +++ b/bun.lock @@ -69,6 +69,7 @@ }, "devDependencies": { "@bahmutov/cypress-esbuild-preprocessor": "^2.2.8", + "@electric-sql/pglite": "^0.5.8", "@mdx-js/mdx": "^3.1.1", "@tailwindcss/postcss": "^4.3.3", "@types/adm-zip": "^0.5.8", diff --git a/docs/dashboard-readonly-views.md b/docs/dashboard-readonly-views.md index 1ca378966..04294ebce 100644 --- a/docs/dashboard-readonly-views.md +++ b/docs/dashboard-readonly-views.md @@ -44,7 +44,9 @@ combinations, not the full Cartesian product of all possible filter values. - Evaluation/reliability: chart-data, date resolution and rolling aggregation. - CollectiveX: selected EP/KV/swap chart and fit helpers. - Submissions/images: existing table, weekly/cumulative and image freshness helpers. -- GPU metrics: shared line/correlation transforms and unsampled statistics. +- GPU metrics: shared line/correlation transforms and stored full-record per-GPU + digests. Live artifacts alone calculate statistics from samples; empty stored + digests remain empty. File/host identity and missing-versus-zero semantics persist. - Video: checksum-verified stored bundles, serving/fidelity selectors and tradeoffs. - Overview/rankings/compare: existing discovery-page assembly and scenario helpers. @@ -61,6 +63,11 @@ OperatorX is feature-gated in navigation and uses page-owned contract. Zoom, theme, axis scale, labels, media playback and report expansion are renderer state. GPU interactive downsampling does not alter returned raw data or statistics. +The GPU statistics table includes startup and warmup for all chips in the selected +series, regardless of chip visibility. It is separate from serving-window power, +J/token and selected-time-window calculations. Run telemetry is DB-first with an +artifact fallback for missing storage; the public view returns private, no-store +responses and preserves upstream 503 failures. Run-specific recognition labels are also presentation-only. Run `35879254139` displays `UMBP MoRI SGLang` through October 9, 2026 in America/New_York @@ -89,6 +96,10 @@ those properties. 私有上传、密钥、提示词、反馈及管理操作不作为公开读取接口。 OperatorX 的入口受功能开关控制,页面使用专属的 `/api/v1/operatorx/*` 接口;目前没有发布 `/api/v1/views/operatorx` 契约。 +GPU 视图优先读取已存遥测,缺少存储数据时回退到产物。全记录统计使用所选文件、 +主机序列的全部芯片摘要,包含启动与 warmup;已有摘要为空时不补算,缺失读数不补零。 +芯片显隐和图表降采样不改变该统计,也不改变 serving-window 或 J/token 的计算口径。 +响应使用 private, no-store,上游 503 保留为错误响应。 测试覆盖契约同步及代表性的筛选行为,并未穷举所有参数组合。生产数据库上的 完整 UI/API 对照仍需集成审查,不能仅凭单元测试宣称已完成。 diff --git a/docs/data-pipeline.md b/docs/data-pipeline.md index ca9a196aa..a7a365103 100644 --- a/docs/data-pipeline.md +++ b/docs/data-pipeline.md @@ -84,23 +84,36 @@ must use the append-only contract below. ### Required Power Publication -Ordinary sweeps that opt into `require-power` upload the producer's -`required-power-sweep-manifest/sweep_manifest.json`. Before any CI ingest upsert, -the app matches its required benchmark rows by recipe fingerprint, concurrency, -and scenario/sequence lengths, then requires valid v2 power and positive energy. -Disaggregated recipes also require both role energy measurements. Identical -per-job and collected artifact copies are allowed; conflicting copies fail. -Matching uses the ingest mapper's canonical identity, including AgentX `users` -precedence over `conc`. After benchmark writes, any required point omitted by a -purge or another filter fails the run; purged data is never restored to satisfy -the declaration. - -The manifest must name the source run and head. A successful earlier attempt of -that same run may supply the scope and retained points when failed jobs are -rerun; ingestion logs both declared and current attempts. Sweeps without this -optional manifest keep historical behavior. The separate PowerX publication -receipt compares ingested 8K/1K and AgentX measurements with the database and -public API after cache invalidation; it does not assert browser rendering. +Required ordinary sweeps upload a versioned +`required-power-sweep-manifest/sweep_manifest.json`. The [shared v2 fixture and +contract](./fixtures/powerx-manifest-v2/README.md) bind the complete required +matrix to source run/head/attempt, point identities, topology, exact measurement +windows, physical node/GPU roles and hashed evidence. Both repositories test the +same bytes. Unversioned required manifests fail closed; optional legacy bundles +retain their existing behavior. + +Artifact preparation validates required evidence before workflow migrations. +Required intent also travels in the dispatch payload, so losing both the manifest +and changelog marker cannot downgrade an ordinary required dispatch. Ingestion +repeats validation before workflow/config upserts, checks purges/backfills before +writing, and projects the resulting published curves from base-table state. It +models the actual latest-attempt, whole-curve and same-image append-only rules. +An unexplained loss of an existing recipe or concurrency point rejects ingestion. +Destructive replacement requires exact old-snapshot and lost-point identities in +the manifest; the ordinary producer supplies no such permission. + +The source run and head must match. A successful earlier attempt of that same run +may supply retained evidence when failed jobs are rerun; ingestion logs declared +and current attempts. Required energy must be finite and positive. Missing, +invalid and measured zero remain different values even though all fail this gate. +Disaggregated deployments require physical evidence for both roles. + +This is pure preflight, not atomic publication. Schema migrations occur after +artifact validation but before the curve check. Concurrent writers and failures +during the existing per-file importer remain a risk; staging plus an atomic, +serialized promotion is the follow-up described in the contract. The separate +PowerX receipt still compares source measurements against the DB and exact-run +API after ingestion. It does not prove latest-curve visibility or browser rendering. ### Append-Only Curve Extensions @@ -526,6 +539,175 @@ Producers (`aggregate_power.py`) annotate every aggregate result row with two op Reads are **permanently tolerant**: `queries/benchmarks.ts` selects the columns as `to_jsonb(br) -> 'power_invalid_reasons'` (and `lb` on the matview branch) rather than bare column references. A bare reference fails during query planning until the next ingest workflow applies the migration, because migrations run in the ingest workflows rather than at Vercel deploy. The key lookup degrades to NULL while the column is missing and is byte-identical once it exists, making deploy order irrelevant. +### PowerX Telemetry Digest (`gpu_metric_*`, migration 016) + +Every single-node benchmark job (`benchmark-tmpl.yml`) samples `nvidia-smi` / +`amd-smi` once per second for its whole lifetime and uploads the CSV as +`gpu_metrics_` next to `bmk_` (agentic jobs: `bmk_agentic_`, +still paired by the bare suffix). The multinode template uploads no `gpu_metrics_` +artifact; its telemetry travels inside `power_audit_` as +`LOGS/power/samples.csv`, one deployment-wide CSV written by srt-slurm's +`dcgm-power` collector (`timestamp_unix, hostname, gpu_index, gpu_uuid, power_w`, +power only). `etl/multinode-power-samples.ts` regroups it per host and the ingest +stores one series per host (`file_name` = `LOGS/power/samples.csv#`), +so multinode and disaggregated points get per-GPU power curves with null clocks, +temperature and utilization. Single-node jobs upload a `power_audit_` bundle too, +so discovery and backfill pairing use it only for a suffix with no `gpu_metrics_` +upload. The PowerX explorer used to download and parse the artifacts from GitHub +on every request and lost them after GitHub's 90-day retention. CI ingest now +digests them at ingest time, in the same step that links server logs: + +- `gpu_metric_series` — one row per (workflow run, artifact, CSV path): vendor, + CSV sha256, sample count, GPU count, recorded window, median cadence, and the + parsed sidecars (`gpu_metrics_context.json`, identity, amd-smi energy counters). +- `gpu_metric_samples` — full-resolution rows, one per (GPU, sample). NVIDIA + fills the six common columns; AMD additionally fills edge/memory temperature, + voltages, FCLK/SOCCLK and multimedia activity. Timestamps are UTC; NVIDIA's + zone-less `YYYY/MM/DD HH:MM:SS.mmm` is interpreted with the context sidecar's + `timestamp_timezone` (the producer writes UTC). +- `gpu_metric_gpu_stats` — per (series, GPU, metric) count/min/max/mean/median/ + p95/p99/stddev computed at ingest; matching algorithm versions avoid rescanning samples. +- `benchmark_result_gpu_metrics` — links each benchmark point to the series that + was recorded while it ran (several series per point for multinode artifacts). + +Ingest is idempotent: the same CSV hash, sidecars, and unique sample count refresh +only the point links when `stats_version` matches `GPU_STATS_VERSION`. An outdated +digest is rebuilt without replacing samples or links. A source change replaces the samples and digest inside one +transaction, including corrected timezone or identity sidecars. Repeated samples +keep the first row per (GPU, timestamp) before computing counts and statistics, +matching the sample table's primary key. Explicitly re-ingesting a run also +repairs older duplicate-inflated counts and digests; `--all` skips runs already +containing series, so target those runs with `--run` or use `--force`. Series are +stored per artifact, not per point: an AgentX per-concurrency job maps to one +point, while older fixed-sequence jobs that swept several concurrencies in one +job share one series across points. Windowing a series to the measured serving +interval is a reader concern; the raw series deliberately includes server +start-up and warm-up so both phases can be inspected. + +Run readers filter artifact prefixes and requested Timeline sources in SQL before +loading samples. The public GPU metrics view loads only its selected file/host +series while retaining the complete artifact-name inventory. Both point and run +readers fetch samples in primary-key pages of at most 10,000 rows, preserving +microsecond cursors and returning the complete selected series in time order. + +The point and run readers assemble a series from separate autocommit statements +(series rows, statistics, samples). After loading, they re-read the series +version key (`csv_sha256`, `sample_count`, `ingested_at`, `stats_version`) and retry the whole +read when a re-ingest committed in between, so one payload never mixes the +statistics of one version with the samples of another. + +`bun run admin:db:backfill-gpu-metrics --all --yes` attaches telemetry for runs +ingested before this migration. The reachable history is bounded by GitHub's +90-day artifact retention (the upload step sets no `retention-days`) because the +GCS backup, which keeps every artifact name, only mirrors `schedule` and `push` +runs on `main`; PR sweeps and manual dispatches — nearly every telemetry-bearing +run — are never copied. Our own GCS reader (`lib/gcs-artifacts.ts`) additionally +ignores everything but `bmk_`/`server_logs_` objects, so widening the mirror's run +filter would also need a reader change before backfill could use it. + +Readers: `/api/gpu-metrics?runId=` serves stored telemetry first, including +`series=power` Timeline buckets reconstructed with retained windows and device +identities. Missing storage falls back to GitHub artifacts. Known-incomplete CSV +fallback must match retained filenames and sample counts; known-incomplete bundles +need exact-source re-ingest. Healthy DB series remain in mixed fallback responses. +Database failures return `503 DATABASE_UNAVAILABLE`; incomplete storage without +usable fallback returns `503 STORED_TELEMETRY_INCOMPLETE` with re-ingest guidance. +Timeline requests use read-only POST with sorted validation basenames in a `sources` +JSON body. Stored coverage of those identities permits an artifact-independent response; +missing siblings, including other windows in the same bundle, use source-level DB-first +merging. Unavailable artifacts leave healthy DB traces readable with explicit missing +sources. `sourceCoverage` describes only the requested identities; legacy GET reports +coverage unknown. Plain CSV fallback applies the adjacent context timezone just like +ingest and bundle reads. Raw multi-file artifacts retain separate file/host series. +Successful reads and storage errors use no-store. + +AgentX nested validation documents are matched to the exact root result and retained +window before receiving a canonical validation filename alias. Original path, result +filename and validation hash remain in stored sidecars. CI attaches recovered source +and window metadata before benchmark publication/upsert; targeted telemetry re-ingest +fills only NULL provenance for a unique explicit run/result/concurrency match, even +when samples are unchanged. Metrics and validity remain untouched. Benchmark metadata +repair additionally needs the existing materialized-view refresh and benchmark-cache +invalidation before the UI planner can discover the restored Timeline source. + +`/api/v1/gpu-metrics-point?id=` powers the PowerX point-detail tab. Every request +checks the current DB revision before reading its Blob payload cache. Sidecar repairs, +new point links and shared-series changes therefore select fresh payloads without a +manual purge. Success, missing-point and error responses use no-store; missing data +is 404 and database failures remain errors. Cache-write failures log a warning and +serve the fresh uncached result; the next request retries cache population. + +The public `/api/v1/views/gpu-metrics` projection and full-record UI table use the +same per-GPU statistics digest for the selected file/host series. Current-version +empty or missing metric digests remain empty. Unversioned or outdated digests +are recomputed read-only from retained DB samples with the shared ingest algorithm. +Incomplete retained samples leave statistics empty while preserving raw data for +the existing source-gap recovery path; stats-only writes fail until the source is repaired. +Statistics include startup and warmup, retain measured zero, and exclude missing +readings per metric after first-wins timestamp/GPU deduplication. Mean is +sample-weighted, percentiles interpolate at `p * (N - 1)`, and standard deviation +divides by `N`. GPU visibility and chart downsampling do not alter this population. +Serving-window power, J/token and selected-time-window calculations remain separate. + +中文:历史遥测和 Timeline 优先读取数据库;缺少存储数据时回退到 GitHub 产物, +文件、主机与 GPU 的身份保持独立。数据库故障返回 503;已知存储不完整且无法恢复时, +返回带定向重新入库提示的 503。Timeline 通过只读 POST 传入所需来源标识;缺失的兄弟曲线 +按来源补齐,GitHub 不可用时仍返回健康的 DB 曲线,并显式标出缺失来源。旧的 GET +没有预期清单,覆盖状态为 unknown。sourceCoverage 仅描述本次请求,不代表整个 run +的完整性;普通 CSV 与 bundle、ingest 使用相同的 context 时区。 +点详情每次读取先核对数据库版本,修正 sidecar 或共享关联后 +无需手动清缓存;响应均为 no-store。全记录统计使用已存摘要,包含启动与 warmup; +缺失读数不补零;当前版本的摘要为空时保留为空,旧版本或无版本摘要从 DB 样本只读重算。它与 serving-window 功率、J/token 和 +用户所选时间窗口的统计分别处理。 + +### Full-record statistics upgrades (migration 017) + +`GPU_STATS_VERSION` identifies the full-record digest algorithm, independently +of source hashes and serving-window power schema versions. Bump it whenever +metric populations, mappings, units or statistical definitions change. Migration +017 marks existing series as version 0 (unknown); it does not certify old digests. +Readers also treat a not-yet-migrated column as version 0, allowing app deployment +before the writer migration. Read fallback does not write to the database. + +Point-cache revisions include both the deployed algorithm version and the stored +version, so an algorithm deployment and subsequent digest repair each bypass old +payloads for every linked point. Browser responses remain no-store; an already +open tab receives the new result on its existing refetch/focus path. + +After applying migration 017 to the explicitly selected DB, use the existing +backfill entry point (commands below are operator instructions, not automatic writes): + +```bash +bun run admin:db:backfill-gpu-metrics --stats-only --all --dry-run +bun run admin:db:backfill-gpu-metrics --stats-only --run --attempt --artifact --dry-run +bun run admin:db:backfill-gpu-metrics --stats-only --run --attempt --artifact --yes +``` + +This mode uses DB samples only: no GitHub, artifact-retention cutoff, GPU work, +benchmark publication, or cache purge. Unspecified attempts include all stored +attempts; `--limit` counts series. Each series is locked and its digest/version +replaced in one transaction. Current versions are no-ops, failures retain the old +complete digest and print a scoped retry command with a nonzero exit status. +Missing samples require source re-ingest. Successful retries skip completed series. +DB `real` columns round to float32; comparisons with original CSV computations +should allow float32 tolerance (e.g. 1e-5 relative), not require bit equality. + +Before production repair, retain a DB snapshot of affected series metadata and +digests. Rollback stops the repair worker and restores the prior app/ingest build; +leave the additive column in place. A prior version-aware build recomputes from +samples when versions differ. To restore pre-versioned software and old stored +results, restore the saved digest plus version together while writers are stopped, +then invalidate the legacy cache. Raw samples and links are never changed by +stats-only repair. Parser/timezone/identity fixes still require source re-ingest. + +中文:统计版本独立于数据 revision 和 serving-window schema。迁移先把旧摘要标记为 +未知版本;读取时可从 DB 样本重算,但不写库。`--stats-only` 支持按 run、attempt、artifact +定向持久修复,也可覆盖全部保留历史,不依赖 GitHub 产物。每个 series 在事务内更新摘要与 +版本,失败保留旧结果并输出重试命令;成功后再次执行为 no-op。部署算法或修复摘要都会 +改变点详情缓存键,页面下次 refetch 时生效。生产修复前保留快照,回滚保留新增列;旧软件 +需要旧摘要时,停写后成对恢复摘要与版本,并清除旧缓存。此过程不修改 samples、关联或 +serving-window 指标。 + ### PowerX publication receipts The normal CI importer writes `POWER_PUBLICATION_MANIFEST` when configured. Each diff --git a/docs/fixtures/powerx-manifest-v2/README.md b/docs/fixtures/powerx-manifest-v2/README.md new file mode 100644 index 000000000..0075b4a13 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/README.md @@ -0,0 +1,101 @@ +# PowerX source manifest v2 + +This small, synthetic one-GPU AgentX bundle is copied byte-for-byte between +InferenceX and InferenceX-app. It tests the file contract. It is **not** a GPU +capture, hardware qualification or publication receipt. + +`artifacts/required-power-sweep-manifest/sweep_manifest.json` is the source +manifest. Evidence paths are relative to `artifacts/`, include their GitHub +artifact directory, and have SHA-256 hashes of the exact retained bytes. + +## Contract + +- `schema-version: 2` versions publication independently of measurement-row + `power_metric_schema_version: 2`. Unsupported or unversioned required bundles + fail closed; legacy optional bundles remain optional. +- `run-id`, `run-attempt`, and `head` name the tested source. A later rerun may + retain successful earlier-attempt evidence from the same run and head. The + consumer compares these fields with GitHub source metadata, including on reuse. +- `matrix` retains the ordinary planner declaration. Every required, + non-evaluation matrix point must occur exactly once in `points`. +- `config_key` is the matrix entry's `exp-name`. `identity` binds model, hardware, + framework, precision, recipe fingerprint, workload/sequence and concurrency. + AgentX uses `agentic_traces` with null `isl`/`osl`; fixed sequence uses + `single_turn`. Full recipe details remain in the matrix and hashed aggregate. +- `topology` declares aggregate versus P/D, single versus multi-node and physical + GPU counts. `devices` binds actual nodes and GPU UUIDs to aggregate, prefill or + decode roles with positive per-device energy. Both P/D roles are mandatory for + disaggregated deployments. A local GPU index alone is insufficient. +- `measurement_window` contains finite Unix-second boundaries, with end after + start. The evidence must describe the same window. +- `artifacts` contains confined relative paths, exact hashes and + `validation_state: "valid"`. Missing files, path escapes, invalid verdicts, + mismatched identities and invalid power reject required publication. Missing + and invalid values remain distinct from numeric zero. Required energy and + derived power must be strictly positive; measured zero fails that gate. + +## Publication intent + +Normal manifests emit `publication: {"mode":"incremental","replacement_scope":[]}`. +Incremental means **no authorized loss**; it does not change existing whole-curve +selection or automatically enable append-only. A complete refresh can pass by +retaining all stable point identities. Recipe/image fingerprint changes may need +a reviewed exact replacement. + +`mode: "replacement"` permits only explicit entries containing: + +- `curve_scope`: the canonical logical curve JSON identity; +- `previous_snapshot_workflow_run_id`: the exact database snapshot superseded; +- `removed_point_identities`: the exact set of lost stable identities. + +No wildcard, stale snapshot or broader permission is implied. The producer's +ordinary path emits no destructive authorization. Authoring authorization needs +explicitly reviewed replacement intent and current snapshot evidence. + +AgentX scope groups model/hardware/framework/precision/workload; AGG/P/D, +parallelism, offload and recipe variants are points in that curve. Fixed-sequence +scopes retain existing spec/disagg/offload boundaries. The consumer models latest +attempts, whole snapshots and same-image append-only inheritance before its first +publication write. Row counts or a stale materialized view cannot establish safety. + +## Remaining publication boundary + +Artifact validation runs before workflow migrations. Curve preflight runs before +workflow/config/benchmark upserts. It is a pure preflight, not an atomic commit: +concurrent writers and failures after the first write can still cause partial +publication. Follow-up work must stage the complete run, recheck the base-table +curve under a target-database publication lock, then atomically promote metadata +and benchmark rows. Sidecar preparation belongs outside that transaction. +Post-write DB/API checks remain necessary; exact-run equality alone does not prove +latest-curve visibility. + +## Read-only fixture check in InferenceX-app + +From the repository root: + +```sh +bun packages/db/src/preflight-power-publication.ts \ + docs/fixtures/powerx-manifest-v2/artifacts 123 1 \ + bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb +``` + +This opens no database connection, dispatches no work and writes no files. It +validates artifacts only; focused curve tests model previous published state. + +## 中文说明 + +此目录是一份最小的合成 AgentX 样例,两个仓库保留完全相同的文件字节, +用于测试 producer 与 consumer 的契约;它不是实机采集或生产发布证据。 + +manifest schema v2 记录来源 run、attempt、测试 SHA、完整 matrix、逐点身份、 +测量窗口、实际 node/GPU UUID/角色及产物哈希。必需功耗路径拒绝缺失、损坏、 +不兼容或能量非正的证据;P/D 必须同时覆盖 prefill 和 decode。历史可选功耗 +路径保留兼容行为。数值 0 不会被改写为“缺失”,但不能通过正能量门槛。 + +默认 incremental 只表示没有授权丢点,不会改变数据库整条曲线替换的语义。 +有意替换必须精确指定旧快照及全部移除点,不能使用通配或过期授权。AgentX +的 AGG、P/D、offload 和配方属于同一条逻辑曲线,校验时必须一起比较。 + +产物校验先于 migration;曲线校验先于发布数据写入。当前实现不是原子发布, +仍需后续 staging、发布锁和事务性 promote 来处理并发及中途写入失败。 +DB/API 写后校验仍保留,exactRun 校验不能单独证明 latest 曲线完整可见。 diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics.csv b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics.csv new file mode 100644 index 000000000..d39b2974a --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics.csv @@ -0,0 +1,3 @@ +timestamp, index, power.draw [W] +2023/11/14 22:13:20.000, 0, 500 +2023/11/14 22:13:22.000, 0, 500 diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics_identity.csv b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics_identity.csv new file mode 100644 index 000000000..139b5a261 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/gpu_metrics_identity.csv @@ -0,0 +1,2 @@ +index, uuid, pci.bus_id, name, driver_version +0, GPU-golden, 00000000:01:00.0, NVIDIA H100, 000.00 diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_node.txt b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_node.txt new file mode 100644 index 000000000..fb271d435 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_node.txt @@ -0,0 +1 @@ +golden-node diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_validation.json b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_validation.json new file mode 100644 index 000000000..42d164cb2 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/agentic_golden/power_validation.json @@ -0,0 +1,21 @@ +{ + "schema_version": 1, + "power_valid": true, + "reasons": [], + "benchmark_window": { + "start_time_unix": 1700000000, + "end_time_unix": 1700000002 + }, + "expected_gpu_count": 1, + "observed_gpu_count": 1, + "observed_gpu_ids": ["0"], + "per_gpu_energy_j": { + "0": 1000 + }, + "metrics": { + "avg_power_w": 500, + "avg_total_gpu_power_w": 500, + "total_gpu_energy_j": 1000, + "joules_per_output_token": 2 + } +} diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/bmk_agentic_golden/agg.json b/docs/fixtures/powerx-manifest-v2/artifacts/bmk_agentic_golden/agg.json new file mode 100644 index 000000000..98df7c14b --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/bmk_agentic_golden/agg.json @@ -0,0 +1,25 @@ +{ + "infmax_model_prefix": "qwen3.5", + "hw": "h100", + "framework": "sglang", + "precision": "fp8", + "recipe_fingerprint": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "scenario_type": "agentic-coding", + "conc": 1, + "users": 1, + "tp": 1, + "ep": 1, + "dp_attention": false, + "disagg": false, + "is_multinode": false, + "num_gpus": 1, + "image": "example/serving:golden", + "power_valid": 1, + "power_metric_schema_version": 2, + "avg_power_w": 500, + "avg_total_gpu_power_w": 500, + "total_gpu_energy_j": 1000, + "joules_per_output_token": 2, + "output_tput_per_gpu": 250, + "median_intvty": 50 +} diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/changelog-metadata/changelog_metadata.json b/docs/fixtures/powerx-manifest-v2/artifacts/changelog-metadata/changelog_metadata.json new file mode 100644 index 000000000..af53bb83d --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/changelog-metadata/changelog_metadata.json @@ -0,0 +1,4 @@ +{ + "require-power": true, + "entries": [] +} diff --git a/docs/fixtures/powerx-manifest-v2/artifacts/required-power-sweep-manifest/sweep_manifest.json b/docs/fixtures/powerx-manifest-v2/artifacts/required-power-sweep-manifest/sweep_manifest.json new file mode 100644 index 000000000..f51df0546 --- /dev/null +++ b/docs/fixtures/powerx-manifest-v2/artifacts/required-power-sweep-manifest/sweep_manifest.json @@ -0,0 +1,98 @@ +{ + "schema-version": 2, + "head": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb", + "run-id": 123, + "run-attempt": 1, + "full-sweep": false, + "matrix": { + "single_node": { + "agentic": [ + { + "exp-name": "golden", + "model-prefix": "qwen3.5", + "model": "Qwen/Qwen3.5", + "runner": "h100", + "framework": "sglang", + "precision": "fp8", + "recipe-fingerprint": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "conc": [1], + "require-power": true, + "scenario-type": "agentic-coding", + "disagg": false, + "num-gpus": 1, + "tp": 1, + "ep": 1, + "image": "example/serving:golden" + } + ] + }, + "multi_node": {} + }, + "publication": { + "mode": "incremental", + "replacement_scope": [] + }, + "points": [ + { + "config_key": "golden", + "identity": { + "model": "qwen3.5", + "hardware": "h100", + "framework": "sglang", + "precision": "fp8", + "recipe_fingerprint": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa", + "benchmark_type": "agentic_traces", + "isl": null, + "osl": null, + "concurrency": 1 + }, + "topology": { + "disagg": false, + "is_multinode": false, + "num_gpus": 1, + "tp": 1, + "ep": 1, + "dp_attention": false + }, + "measurement_window": { + "start_time_unix": 1700000000, + "end_time_unix": 1700000002 + }, + "devices": [ + { + "node": "golden-node", + "gpu_uuid": "GPU-golden", + "role": "aggregate", + "energy_j": 1000 + } + ], + "artifacts": [ + { + "path": "agentic_golden/gpu_metrics.csv", + "sha256": "f42dee832f8d837142be9d4dc5fbbf8462c02846f06e66edfce659b8b20943db", + "validation_state": "valid" + }, + { + "path": "agentic_golden/gpu_metrics_identity.csv", + "sha256": "a4364a0d4050078b9cba24c7e695fbed658fabb1d48e61a3da8f61b8ec873627", + "validation_state": "valid" + }, + { + "path": "agentic_golden/power_node.txt", + "sha256": "114aa18bee3688c0e6040020ad0230472645541372b88a54253116fd9c2148e6", + "validation_state": "valid" + }, + { + "path": "agentic_golden/power_validation.json", + "sha256": "c00687e7239fd172aa15f60c8d967294a0fa623984d0eaeeb5222424b6335d94", + "validation_state": "valid" + }, + { + "path": "bmk_agentic_golden/agg.json", + "sha256": "56e3377ae12402eec61e25e2f9a0480ab228048a29d32c35cb8d438d50ea2faa", + "validation_state": "valid" + } + ] + } + ] +} diff --git a/docs/fixtures/powerx-reingest/README.md b/docs/fixtures/powerx-reingest/README.md new file mode 100644 index 000000000..6df4caba2 --- /dev/null +++ b/docs/fixtures/powerx-reingest/README.md @@ -0,0 +1,34 @@ +# Retained telemetry for re-ingest recovery + +`nvidia.csv` contains unchanged readings from the public InferenceX B200/Qwen3.5 +run [34175132645, attempt 1](https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34175132645/attempts/1). +Tests read these committed bytes; no artifact download is required. + +Retain original line 1 and lines 2898–2921, joined with LF and a final LF: three +consecutive samples of all eight GPUs, 24 rows / 1,901 bytes. Headers, units, +timestamps and values are unchanged. The CSV contains no hostnames or UUIDs, so +no identity anonymization was needed. The original collector version and timezone +are unknown. + +The UTC context, collector labels, identity sidecars, local benchmark rows and +repair offsets used by tests are **constructed test inputs**. They do not describe +a correction to the original run. No original energy/quality claim is attached to +this cropped fragment. It proves retained-format parsing and recovery through the +consumer, not GPU collection reliability or serving-window qualification. + +Fixed checks come directly from the retained rows: GPU 0 has +`380.42, 336.61, 337.18 W`; with the explicit UTC test context the range is +`2026-09-08T07:20:19.279Z`–`2026-09-08T07:20:21.279Z`. Its sample mean is +`351.4033333333333 W`. Production parsers/statistics functions must not generate +test expectations. + +Run the recovery check from `packages/app`: + +```sh +bun node_modules/vitest/vitest.mjs run src/app/api/v1/gpu-metrics-point/route.test.ts +``` + +The test uses PGlite, the production parser/ingest/query/cache/point handler, and +the real Blob SDK with a local HTTP transport. It checks live revisions for +successful and missing responses, then refreshes a missing point, shared links and +timezone/identity repairs without purging. diff --git a/docs/fixtures/powerx-reingest/nvidia.csv b/docs/fixtures/powerx-reingest/nvidia.csv new file mode 100644 index 000000000..62113072c --- /dev/null +++ b/docs/fixtures/powerx-reingest/nvidia.csv @@ -0,0 +1,25 @@ +timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%] +2026/09/08 07:20:19.279, 0, 380.42 W, 37, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:19.279, 1, 392.58 W, 43, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:19.279, 2, 375.45 W, 36, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:19.279, 3, 384.71 W, 42, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:19.279, 4, 379.91 W, 36, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:19.279, 5, 388.18 W, 42, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:19.279, 6, 374.34 W, 37, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:19.279, 7, 379.36 W, 41, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:20.279, 0, 336.61 W, 37, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:20.279, 1, 346.06 W, 43, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:20.279, 2, 333.34 W, 36, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:20.279, 3, 344.15 W, 42, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:20.279, 4, 337.25 W, 36, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:20.279, 5, 342.84 W, 42, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:20.279, 6, 335.00 W, 37, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:20.279, 7, 335.78 W, 41, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:21.279, 0, 337.18 W, 37, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:21.279, 1, 345.05 W, 43, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:21.279, 2, 340.15 W, 36, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:21.279, 3, 344.63 W, 42, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:21.279, 4, 337.94 W, 36, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:21.279, 5, 342.85 W, 42, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:21.279, 6, 334.88 W, 37, 1965 MHz, 3996 MHz, 100 %, 9 % +2026/09/08 07:20:21.279, 7, 336.11 W, 41, 1965 MHz, 3996 MHz, 100 %, 9 % diff --git a/docs/index.md b/docs/index.md index 812ef351e..c67e877ae 100644 --- a/docs/index.md +++ b/docs/index.md @@ -9,6 +9,7 @@ Design rationale and non-obvious conventions. See [CLAUDE.md](../CLAUDE.md) for - [API Skill Examples](./inferencex-api-examples.md) — Install the public skill, query benchmarks, export measured PowerX data, and explain empty results - [PowerX System Power](./powerx-system-power.md) — Pinned chassis model, measured-input guards, assumptions, and reproducible article exports +- [PowerX Persistence and Recovery](./powerx-persistence-recovery.md) — Telemetry receipts, migration prerequisites, and targeted repair - [API Skill Releases](./inferencex-skills-release.md) — Prepare an immutable package, verify clean installations and agent exports, and publish through the package-specific workflow - [API Skill Discovery](./inferencex-skills-discovery.md) — Accept or reject implicit skill discovery in fresh Codex and Claude Code projects - [CLI Compatibility](./inferencex-cli-compatibility.md) — Contract 1 schemas, bundle and exit semantics, platform scope, and legacy migration diff --git a/docs/powerx-persistence-recovery.md b/docs/powerx-persistence-recovery.md new file mode 100644 index 000000000..898d5ace7 --- /dev/null +++ b/docs/powerx-persistence-recovery.md @@ -0,0 +1,307 @@ +# PowerX persistence and repair + +PowerX point detail, the run explorer and Power Timeline read migration-016 telemetry +from the database. Timeline applies the same prefix selection, validation-window cuts, +60-second padding and one-second per-device means as the artifact path. Stored samples +retain UTC timestamps, original units and separate host-local GPU identities. GitHub +remains the fallback for telemetry that has not been stored. Database failures return an +error, not an empty result or an artifact fallback. + +Timeline sends a read-only `POST /api/gpu-metrics?runId=RUN_ID&series=power&prefix=PREFIX` +with JSON `{ "sources": ["power_validation_RESULT_FILENAME.json"] }`. The planner sorts +and deduplicates validation basenames, and includes them in the React Query key. The body +accepts 1–1000 identities, each RESULT_FILENAME up to 200 ASCII letters/digits/`.`/`_`/`-`, +matching `prefix` when supplied; invalid input returns 400 and bodies above 256 KiB return 413. POST avoids putting a whole run's identity list in a URL. GET remains compatible with +the raw explorer and existing callers. + +Only requested identities absent from healthy DB series trigger artifact discovery. Merging +prefers DB by validation identity, so a live sibling window in the same bundle is retained. +`sourceCoverage: { status, missingSources }` reports coverage of the requested identities: +`complete` or `incomplete` for POST, `unknown` for GET without an expectation list. It is +not a full-run/sweep or raw-sample completeness receipt. An unavailable GitHub token, listing +or download keeps healthy requested DB series readable with incomplete coverage; it cannot +hide known-incomplete stored telemetry or database failures. Unrelated stored artifacts +outside the requested identities do not block that read. + +Live ordinary CSVs and bundle CSVs normalize NVIDIA wall-clock timestamps to ISO UTC with +the ingest parser, so live and stored reads agree. Adjacent collector context supplies the +offset; missing or UTC context means zero. Context must be in the same ZIP directory as +the CSV; host directories never share offsets. When multiple context candidates exist, both +ingest and live reads select the first valid object in code-unit filename order, independent +of archive or filesystem listing order. ISO/AMD timestamps pass through unchanged. + +Both paths keep the first row for a device/timestamp, matching the existing ingest +deduplication, before averaging distinct samples within a bucket. A discovered unreadable +host CSV alongside readable CSVs fails telemetry preparation before any series is written; +it cannot disappear from the stored inventory and falsely establish complete coverage. + +New ingests preserve validation documents, bundle manifests and expected file/sample +inventories in the existing series sidecars. Older ingests +can recover validation source/window from linked benchmark `power_audit` provenance. +Information never retained, such as historical validation role overrides, cannot be +reconstructed: recover the exact original artifact and re-ingest it. An incomplete +stored bundle must not silently become a smaller deployment's average. Incomplete stored +telemetry can use live CSV fallback only when every retained file and deduplicated sample +count is present. Missing/header-only host files, truncated data, and a second incomplete +artifact must not disappear behind another artifact's successful recovery. Complete stored +artifacts remain in a fallback response even if their GitHub artifacts have expired. +A known-incomplete power bundle needs re-ingest because the downloaded window cuts do not +prove its full original host/sample inventory. Un-ingested bundles still use normal artifact +fallback. Unverifiable recovery returns `503 STORED_TELEMETRY_INCOMPLETE` with the artifact +and a targeted re-ingest action. DB query failures return `503 DATABASE_UNAVAILABLE` and do +not attempt artifact fallback. + +The live raw explorer also keeps each CSV as a separate series, matching the database +reader. Multi-file artifact labels include the CSV path, so host-local GPU 0 values cannot +be merged into one device. Single-file labels remain unchanged. + +AgentX artifacts may retain their validation at +`LOGS/agentic/conc_N/power_validation.json`. Both stored and artifact readers recognize +this layout only when the exact root result filename, its `conc`, and the retained +`LOGS/power/windows/agentic_power_concurrency_N.json` agree with the validation's +selected window. Missing, malformed or mismatched evidence does not create a window. +The normalized source is `power_validation_.json`; it is an alias, +not an invented top-level file. Stored validation sidecars retain `validation_path`, +`result_file` and the SHA-256 of the original validation bytes. Legacy top-level +validation documents take precedence, and power-validity and role values stay unchanged. + +Normal CI derives missing AgentX benchmark source/window metadata before publication +and benchmark upsert, retaining it across aggregate copies of the same full point +identity. Targeted telemetry re-ingest also fills SQL-NULL `power_audit` after all hosts +are stored and linked, including sample no-ops. It requires one unambiguous AgentX point +within the caller's explicit run/result IDs with matching concurrency; it neither +overwrites non-null provenance nor changes metrics, workers or validity. The ingest +result reports actual `metadataUpdatedBenchmarkResultIds`. Ordinary benchmark reads +also require the existing `latest_benchmarks` refresh and benchmark cache invalidation; +point/Timeline revision changes alone do not refresh the UI's benchmark source list. + +## Cache and browser recovery + +`/api/v1/gpu-metrics-point?id=N` checks a live database revision before reading its existing +Blob cache. The revision includes linked series, ingest time, CSV hash, sidecars and shared +point links/audits. The payload stays in Blob; the revision query reads only metadata. +Same-input ingest leaves the revision unchanged. A sidecar correction, series replacement, +new point link or linked audit correction selects a new cache entry, including for every +point sharing the series. Missing points are not negatively cached. A re-ingest that +commits while the reader is between its statements changes the series version key +(`csv_sha256`, `sample_count`, `ingested_at`), which the reader re-checks after loading +and retries, so a payload never mixes statistics and samples from two versions. + +Point and Timeline responses use `Cache-Control: no-store`, so CDN/browser response caches +cannot skip the database check. Client queries revalidate on mount and window focus. +An already open page refreshes when revisited/refocused or reloaded; there is no continuous +polling. A failed Blob write logs `serving uncached result`, returns fresh DB data, and is +retried on the next read. A database failure remains an API error even if an old Blob exists. +Old cache generations remain until the existing prefix cleanup runs; repair does not require +a manual PowerX cache purge. General benchmark cache invalidation still follows normal CI. + +## Coverage receipts + +The existing `power-publication.json` gains an optional `telemetry` receipt, keyed by run, +attempt and stable benchmark-point identity. Its stages distinguish expected benchmark +attachments, produced artifacts, stored series/samples, point links and actual API readback. +Normal CI builds expectations from benchmark rows before telemetry discovery. Backfill +also inspects benchmark siblings whose telemetry artifact is absent. Unreadable benchmark +siblings are recorded in `expectationErrors`; known identities still proceed independently. + +`plannedPointCount: null` explicitly means the full planned sweep is unknown. Attachment +coverage over available benchmark rows does **not** prove full sweep coverage. Point, +artifact, series and sample counts are separate: a bundle can contain multiple host series +and multiple points can share those series. Receipt entries list missing identities, +reasons, and exact run/attempt/artifact recovery targets. DB success leaves API status +`unknown`; only `verify-power-publication.ts` advances it after an actual HTTP read. +Telemetry failures remain isolated from benchmark publication failure policy. + +The expectation boundary is the successfully mapped, non-purged benchmark rows +selected for ingestion, plus persisted benchmark identities recovered by the +receipt query. It is not every raw result or every planned job. Mapping/preflight +errors remain in the existing ingest diagnostics. Unreadable/unmappable benchmark +rows and failed config resolution make `expectedSource` unknown while retaining +known point identities; intentionally failed or purged benchmarks remain excluded. +The telemetry receipt does not +reconstruct the planned matrix, even when a separate required-power manifest is +available; `plannedPointCount` stays null. An unreadable backfill benchmark sibling +sets `expectationErrors` and makes `expectedPoints` unknown, including when its +telemetry pair exists. +Backfill preserves mapped points alongside per-row diagnostics for unmappable rows; +multiple row errors in one benchmark artifact count as one failed artifact. Every +mapped point in a correction must match a persisted benchmark before telemetry +ingest can clear that artifact's recovery failure. An uploaded correction with an +unmatched point is not a successful repair, even when older data remains readable. +Later refreshes retain unresolved expectation errors until that exact sibling's +identities can actually be recovered; an expired sibling cannot disappear from +the unknown denominator. +This also applies on the first receipt: a selected expired benchmark sibling records +an expectation error without attempting a download. Superseded retries and unrelated +targets remain excluded by the existing logical-name and artifact filters. +Known failed benchmark rows remain excluded, matching normal CI. + +Historical backfill retains the resolver's exact-first, unique-fallback offload matching. +A proven fallback uses the persisted point's offload identity before receipt counting +and recovery checks, including benchmark siblings whose telemetry is absent. Repeating +the recovery removes only the corresponding old null-ID phantom in that artifact's +scope; real on/off points remain distinct. Conflicting artifact observations or prior +artifact scope stay visible with an unknown denominator rather than overwriting errors. + +Normal CI deliberately retains successful artifacts from earlier attempts of the +same run/head for `rerun-failed`. Receipt `runAttempt` identifies the persisted +ingest cohort, not the collection attempt of every artifact. Per-artifact collection +attempt is unknown unless independently retained in source provenance. Do not +filter the normal artifact plan to the latest attempt and silently drop successful +points. Explicit `--download .../attempts/N` rejects a different current attempt; +targeted GitHub telemetry recovery uses its stricter current-attempt filter below. + +An empty or wholly unreadable telemetry correction is an ingest failure, even when +older stored samples remain readable. A targeted recovery retains unresolved +`recoveryError` and `recoveryArtifactNames` from other artifacts. Sequential successful +repairs remove only their own artifact names. Older receipts with an unscoped recovery +error require a successful full-run refresh to clear that error; repairing a different +artifact cannot establish recovery. Retained error text describes the failed attempts; +the remaining artifact names identify the outstanding scope. + +| Receipt reason | Targeted action | +| -------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `artifact_missing` | Recover the exact listed sibling from the same source provenance; use retained local bytes if GitHub no longer has that attempt. Do not substitute a newer attempt. | +| `ingest_failed` | Inspect the recorded error, correct the exact artifact or sidecar, then re-ingest it. Empty/unparseable CSVs do not count as a successful correction. | +| `point_link_missing` | Re-ingest the named unchanged artifact for the recorded run/attempt. Existing series/samples remain idempotent and the missing link is restored. | +| `expectationErrors` | Recover the named benchmark sibling first so its point identities can be enumerated; successful telemetry downloads alone do not establish the denominator. | + +Apply the run/attempt/artifact selectors in each point's `recovery` object to the +commands below, merge the same receipt, then run the HTTP verifier. A missing explicit +target or deleted GitHub run reports failure. Neither a successful command nor the +benchmark-level `matched` status establishes telemetry completeness: inspect the +separate counts, unknowns and outstanding errors. + +The stored inventory distinguishes `storage.status` complete/incomplete/unknown. Missing +host files and mismatched sample counts remain gaps even when the surviving payload can be +read. `apiReadablePoints` counts actual successful HTTP reads; `apiCompletePoints` additionally +requires the retained artifact inventory and every point link. Legacy rows without an inventory +remain unknown until recovered; they do not silently count as complete. A failed correction or +receipt recovery error also blocks the complete count, even when older data remains readable. + +## Full-record statistics + +The point detail, run explorer and public `/api/v1/views/gpu-metrics` projection use the +stored per-GPU digest for the existing full-record statistics, with name mappings only. +Units and percentile/stddev definitions are unchanged. +Zero is a value; missing metrics or an empty digest remain missing. Live, un-ingested +artifact data still computes statistics in the browser. These tables include startup and +warmup. They are not serving-window power, energy, or user-selected-window statistics. + +For each file/host series and GPU, the population keeps the first row at each timestamp, +then excludes missing or nonfinite readings separately for each metric. Live CSV parsing +uses the same rule; a missing optional temperature or clock does not discard valid power. +Count is exact, mean is sample-weighted, P50/P95/P99 use linear interpolation at +`p * (N - 1)`, and standard deviation uses the population denominator `N`. Multi-host +GPU indices must remain in separate series. No time weighting, smoothing, chart zoom or +GPU visibility filter changes the full-record table. Window calculations still consume +the selected samples, and serving energy/financial metrics retain their own definitions. +SQL samples and digests use `real` (float32), so compare stored values with an explicit +float32 tolerance rather than the tighter tolerance of the in-memory parser tests. + +## Local verification + +From the repository root: + +```sh +bun install --frozen-lockfile --ignore-scripts +bun run test:unit +bun run typecheck +bun run lint +bun run fmt +bun run check:typography +bun run --cwd packages/app test:unit src/app/api/v1/gpu-metrics-point/route.test.ts +bun run --cwd packages/app test:unit src/app/api/v1/views/gpu-metrics/route.test.ts +bun run --cwd packages/db test:unit src/queries/gpu-metrics-timeline.test.ts src/etl/telemetry-receipt.test.ts +``` + +The ordinary fixture-backed smoke command remains `bun run test:e2e` with an +`E2E_FIXTURES=1` dev server. The full cross-browser matrix remains the repository CI gate. + +## Deployment and targeted data repair (operator review required) + +1. Deploy the reviewed application and ingest/backfill code together. Check the target's + migration ledger and apply pending migrations with `bun run admin:db:migrate --yes` + before running ingest or backfill. This branch adds `016_gpu_metrics.sql` and + `017_gpu_metric_stats_version.sql`; writers require the `stats_version` column from 017. CI ingest workflows migrate automatically, but manual backfill does not. + Do not run migrations or backfill against production merely to inspect a receipt. +2. Retain the original publication receipt, exact artifact bytes, sidecars, source run and + attempt. Snapshot the target run's telemetry series, samples, digests and point links + before correction. Check that the app reads the same database that ingest writes. +3. Inspect a bounded candidate with the existing command, using explicitly selected + target credentials in the operator environment: + + ```sh + bun run admin:db:backfill-gpu-metrics --run RUN_ID --attempt ATTEMPT --dry-run + ``` + +4. Once authorized, repair only the named artifact and merge the existing receipt: + + ```sh + bun run admin:db:backfill-gpu-metrics --run RUN_ID --attempt ATTEMPT \ + --artifact EXACT_ARTIFACT_NAME --receipt /path/power-publication.json --yes + bun packages/db/src/verify-power-publication.ts /path/power-publication.json https://TARGET_ORIGIN + ``` + + Before a backfill that may fill missing AgentX `power_audit`, set + `CACHE_INVALIDATE_URL=https://TARGET_ORIGIN/api/v1/invalidate` and + `CACHE_INVALIDATE_SECRET` (or `INVALIDATE_SECRET`); protected previews also need + `CACHE_PROTECTION_BYPASS_SECRET`. The selected app must use the same DB and its + existing cache namespace/Blob prefix. No new cache scope is introduced. + + The backfill checkpoints NULL-audit AgentX candidates in the existing manifest + before ingest writes. `benchmarkRefresh` records IDs, retained source-derived + audit updates, target endpoint, pending/complete/failed status, check time and + error. DB audits must match that saved source evidence. Only the matching + publication points' NULL audit fields are enriched; original artifact hashes, + benchmark values and unrelated points remain unchanged. Without `--receipt`, the + existing default is `power-publication-RUN_ID-attempt-ATTEMPT.json`. Missing + configuration fails that pair before ingest; configuration errors cannot + silently turn a metadata write into a successful backfill. + + After ingest, refresh runs in this order: `latest_benchmarks` materialized view, + the existing invalidate endpoint, then exact-run benchmark API comparison by ID + and structured `power_audit`. Refresh failure exits nonzero and retains the + responsibility even if the next ingest changes no samples or metadata. + To retry only that phase, without downloading artifacts or rewriting samples: + + ```sh + bun run admin:db:backfill-gpu-metrics --run RUN_ID --attempt ATTEMPT \ + --receipt /path/power-publication.json --refresh-cache-only --yes + ``` + + The receipt's run/attempt, point IDs and endpoint must still match. A checkpoint + interrupted before its metadata UPDATE leaves NULL candidates. They retain a + failed refresh responsibility and require targeted artifact re-ingest; a later + upsert clearing an already-written audit cannot erase that responsibility. + Invalid/string-encoded audits fail explicitly. HTTP calls have a 30-second + timeout; retries use the saved receipt rather than an unrecorded manual purge. + Dry runs never refresh. `complete` covers the recorded IDs and API metadata, + not browser state, energy validity or deployment-wide acceptance. + + An explicit `--run` is rechecked even if some series already exist. Unchanged inputs + are no-ops. The GitHub backfill route accepts only the current source attempt and + filters out earlier attempt artifacts; it refuses a mismatched historical attempt. + Expired or older-attempt bytes require retained artifacts through the existing local + ingest path (`INGEST_ARTIFACTS_PATH` and exact source-run metadata), after operator + review. Never substitute another attempt's bytes. A local sidecar correction must be + applied to the retained artifact tree; re-downloading unchanged GitHub bytes cannot fix it. + The local ingest entry takes `INGEST_RUN_ID`, `INGEST_RUN_ATTEMPT`, + `INGEST_REPO=SemiAnalysisAI/InferenceX`, `INGEST_ARTIFACTS_PATH` and + `POWER_PUBLICATION_MANIFEST`, then `bun run admin:db:ingest:ci`. It also requires + `GITHUB_TOKEN` and the reviewed `DATABASE_WRITE_URL`. Inspect any retained + `reused-ingest-metadata/reuse_source_run.json` first: it overrides run/attempt identity. + +5. Inspect per-point receipt gaps and both point/Timeline APIs, then refocus/reload the + browser. Check expected hosts, GPU IDs, timestamps and digest values. Ingest success + alone is insufficient. No telemetry cache purge is required. + +For rollback, revert application code only if needed; additive sidecars/receipt fields are +compatible with the previous schema. Reverting application code restores its old cache +behavior, so the existing cache invalidation procedure is required in that case. To undo +a data correction, re-ingest the retained original bytes/sidecars against the exact same +run/attempt/point identities, or restore only the snapshotted rows in a reviewed transaction. +Verify links and counts again. Do not delete a shared series to repair one point. + +Local tests do not establish deployment, production repair, expired-artifact recoverability, +or complete published PowerX coverage. diff --git a/package.json b/package.json index ba63caaca..0849c661e 100644 --- a/package.json +++ b/package.json @@ -44,6 +44,7 @@ "admin:db:migrate:collectivex": "bun run --cwd packages/db db:migrate:collectivex", "admin:db:apply-overrides": "bun run --cwd packages/db db:apply-overrides", "admin:db:backfill-full-response-interactivity": "bun run --cwd packages/db db:backfill-full-response-interactivity", + "admin:db:backfill-gpu-metrics": "bun run --cwd packages/db db:backfill-gpu-metrics", "admin:db:backfill-server-log-files": "bun run --cwd packages/db db:backfill-server-log-files", "admin:db:reset": "bun run --cwd packages/db db:reset", "admin:db:verify": "bun run --cwd packages/db db:verify", diff --git a/packages/app/cypress/e2e/blog.cy.ts b/packages/app/cypress/e2e/blog.cy.ts index a527bff8b..b70a927ce 100644 --- a/packages/app/cypress/e2e/blog.cy.ts +++ b/packages/app/cypress/e2e/blog.cy.ts @@ -38,18 +38,13 @@ describe('Blog', () => { }); describe('Blog listing page', () => { - before(() => { - // Stub remote Substack thumbnails with a real 1×1 PNG. A 204 empty body - // can prevent Firefox from firing `window.load` while eager card images - // stay pending, which times out `cy.visit` in before-all. - cy.intercept( - { - method: 'GET', - pathname: '/_next/image', - query: { url: /^https:\/\/substack-post-media\.s3\.amazonaws\.com\// }, - }, - { fixture: '1x1.png', headers: { 'content-type': 'image/png' } }, - ); + beforeEach(() => { + // The listing checks text and links, so stub every optimized thumbnail. + // Waiting on image optimization can keep Firefox from firing `window.load`. + cy.intercept('GET', '**/_next/image?*', { + fixture: '1x1.png', + headers: { 'content-type': 'image/png' }, + }); cy.visit('/blog'); }); diff --git a/packages/app/cypress/e2e/gpu-power.cy.ts b/packages/app/cypress/e2e/gpu-power.cy.ts index 2ff041dd5..eb769267b 100644 --- a/packages/app/cypress/e2e/gpu-power.cy.ts +++ b/packages/app/cypress/e2e/gpu-power.cy.ts @@ -211,9 +211,9 @@ describe('PowerX Chinese route', () => { .should('contain.text', '秒') .and('contain.text', '功耗 (W)'); cy.get('[data-testid="gpu-metrics-chart-svg"]').should('contain.text', '点击数据点固定提示框'); - cy.get('[data-testid="gpu-metrics-chart-svg"] svg .point') - .first() - .trigger('mouseenter', { force: true }); + cy.get('[data-testid="gpu-metrics-chart-svg"] svg .point').first().scrollIntoView(); + // Exercise the supported mobile interaction, not a forced off-screen hover. + cy.get('[data-testid="gpu-metrics-chart-svg"] svg .point').first().click(); cy.get('[data-chart-tooltip]:visible') .should('contain.text', '芯片 0') .and('contain.text', '功耗:'); diff --git a/packages/app/package.json b/packages/app/package.json index 3ee80be32..dbc291ffe 100644 --- a/packages/app/package.json +++ b/packages/app/package.json @@ -83,6 +83,7 @@ }, "devDependencies": { "@bahmutov/cypress-esbuild-preprocessor": "^2.2.8", + "@electric-sql/pglite": "^0.5.8", "@mdx-js/mdx": "^3.1.1", "@tailwindcss/postcss": "^4.3.3", "@types/adm-zip": "^0.5.8", diff --git a/packages/app/scripts/powerx-blob-fixture.ts b/packages/app/scripts/powerx-blob-fixture.ts new file mode 100644 index 000000000..ea283fc10 --- /dev/null +++ b/packages/app/scripts/powerx-blob-fixture.ts @@ -0,0 +1,76 @@ +import { createServer } from 'node:http'; +import type { AddressInfo } from 'node:net'; + +/** Local HTTP transport for the real Blob SDK; never contacts a Blob store. */ +export async function startPowerxBlobFixture(port = 0) { + const objects = new Map(); + const counts = { reads: 0, writes: 0, failedWrites: 0 }; + let rejectWrites = false; + const server = createServer(async (request, response) => { + const url = new URL(request.url!, 'http://localhost'); + const pathname = url.searchParams.get('pathname') ?? url.searchParams.get('url'); + response.setHeader('content-type', 'application/json'); + const fail = (status: number, code: string) => { + response.writeHead(status); + response.end(JSON.stringify({ error: { code, message: code } })); + }; + if (url.pathname === '/body') { + counts.reads++; + const body = objects.get(pathname!); + if (body === undefined) return fail(404, 'not_found'); + response.end(body); + return; + } + if (!pathname) return fail(400, 'bad_request'); + if (request.method === 'PUT') { + counts.writes++; + if (rejectWrites) { + counts.failedWrites++; + request.resume(); + return fail(403, 'forbidden'); + } + const chunks: Buffer[] = []; + for await (const chunk of request) chunks.push(Buffer.from(chunk)); + objects.set(pathname, Buffer.concat(chunks).toString()); + } + if (!objects.has(pathname)) return fail(404, 'not_found'); + const address = server.address() as AddressInfo; + const bodyUrl = `http://127.0.0.1:${address.port}/body?url=${encodeURIComponent(pathname)}`; + response.end( + JSON.stringify({ + url: bodyUrl, + downloadUrl: bodyUrl, + pathname, + size: Buffer.byteLength(objects.get(pathname)!), + contentType: 'application/json', + uploadedAt: new Date().toISOString(), + etag: 'local-fixture', + }), + ); + }); + await new Promise((resolve) => { + server.listen(port, '127.0.0.1', resolve); + }); + const address = server.address() as AddressInfo; + return { + env: { + VERCEL_BLOB_API_URL: `http://127.0.0.1:${address.port}`, + BLOB_READ_WRITE_TOKEN: 'vercel_blob_rw_local_fixture', + BLOB_CACHE_PREFIX: 'powerx-local-acceptance', + }, + objects, + counts, + rejectWrites: (value: boolean) => { + rejectWrites = value; + }, + close: () => + new Promise((resolve, reject) => { + server.closeAllConnections(); + server.close((error) => + error && (error as NodeJS.ErrnoException).code !== 'ERR_SERVER_NOT_RUNNING' + ? reject(error) + : resolve(), + ); + }), + }; +} diff --git a/packages/app/src/app/api/gpu-metrics/route.context.test.ts b/packages/app/src/app/api/gpu-metrics/route.context.test.ts new file mode 100644 index 000000000..1e2b58eeb --- /dev/null +++ b/packages/app/src/app/api/gpu-metrics/route.context.test.ts @@ -0,0 +1,109 @@ +import AdmZip from 'adm-zip'; +import { NextRequest } from 'next/server'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; + +const { readStored } = vi.hoisted(() => ({ readStored: vi.fn() })); +vi.mock('@semianalysisai/inferencex-db/connection', () => ({ getDb: () => ({}) })); +vi.mock('@semianalysisai/inferencex-db/queries/gpu-metrics', () => ({ + getGpuMetricsForRun: readStored, +})); + +import { GET } from './route'; + +const HEADER = + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]'; +const CSV = `${HEADER}\n2026/09/21 00:00:00.000, 0, 100 W, 40, 1500, 2000, 90, 80\n2026/09/21 00:00:01.000, 0, 200 W, 41, 1500, 2000, 90, 80\n`; + +beforeEach(() => { + readStored.mockReset().mockResolvedValue(null); + vi.stubEnv('GITHUB_TOKEN', 'controlled-not-a-real-token'); + vi.stubEnv('DATABASE_READONLY_URL', 'postgresql://controlled.invalid/never-contacted'); +}); + +afterEach(() => { + vi.unstubAllGlobals(); + vi.unstubAllEnvs(); +}); + +function liveArtifact(files: Record) { + const zip = new AdmZip(); + for (const [entry, contents] of Object.entries(files)) zip.addFile(entry, Buffer.from(contents)); + const bytes = zip.toBuffer(); + vi.stubGlobal( + 'fetch', + vi.fn((input: unknown) => { + const url = String(input); + if (url.includes('/artifacts?')) { + return Promise.resolve( + Response.json({ + artifacts: [ + { + id: 2, + name: 'gpu_metrics_probe_live', + archive_download_url: 'https://example.test/dl/live', + }, + ], + }), + ); + } + if (url === 'https://example.test/dl/live') { + return Promise.resolve( + new Response(new Uint8Array(bytes), { + headers: { 'Content-Length': String(bytes.length) }, + }), + ); + } + if (url.endsWith('/actions/runs/12345')) { + return Promise.resolve( + Response.json({ + id: 12345, + name: 'Controlled context fixture', + head_branch: 'test', + head_sha: 'fixture', + created_at: '2026-09-21T00:00:00Z', + html_url: 'https://example.test/runs/12345', + conclusion: 'success', + status: 'completed', + }), + ); + } + throw new Error(`Unexpected network request: ${url}`); + }), + ); +} + +async function readRoute() { + const response = await GET(new NextRequest('http://localhost/api/gpu-metrics?runId=12345')); + expect(response.status).toBe(200); + return response.json(); +} + +describe('GET /api/gpu-metrics collector context', () => { + it('uses each host directory context and skips malformed neighboring context files', async () => { + liveArtifact({ + 'gpu_metrics_context.json': '{"timestamp_timezone":"+12:00"}', + 'host-a/gpu_metrics.csv': CSV, + 'host-a/gpu_metrics_bad_context.json': '{bad json', + 'host-a/gpu_metrics_context.json': '{"timestamp_timezone":"-07:00"}', + 'host-b/gpu_metrics.csv': CSV, + 'host-b/gpu_metrics_context.json': '{"timestamp_timezone":"+05:30"}', + }); + const raw = await readRoute(); + expect(raw.artifacts).toMatchObject([ + { + name: 'gpu_metrics_probe_live/host-a/gpu_metrics.csv', + data: [ + { timestamp: '2026-09-21T07:00:00.000Z' }, + { timestamp: '2026-09-21T07:00:01.000Z' }, + ], + }, + { + name: 'gpu_metrics_probe_live/host-b/gpu_metrics.csv', + data: [ + { timestamp: '2026-09-20T18:30:00.000Z' }, + { timestamp: '2026-09-20T18:30:01.000Z' }, + ], + }, + ]); + }); +}); diff --git a/packages/app/src/app/api/gpu-metrics/route.sources.test.ts b/packages/app/src/app/api/gpu-metrics/route.sources.test.ts new file mode 100644 index 000000000..afe830736 --- /dev/null +++ b/packages/app/src/app/api/gpu-metrics/route.sources.test.ts @@ -0,0 +1,120 @@ +import { NextRequest } from 'next/server'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; + +import type { GpuMetricSeries } from '@semianalysisai/inferencex-db/queries/gpu-metrics'; +import { parseCsvData } from '@/components/gpu-power/types'; + +const { readRun } = vi.hoisted(() => ({ readRun: vi.fn() })); +vi.mock('@semianalysisai/inferencex-db/connection', () => ({ + getDb: () => ({}), +})); +vi.mock('@semianalysisai/inferencex-db/queries/gpu-metrics', () => ({ + getGpuMetricsForRun: readRun, +})); + +import { POST } from './route'; + +const RUN_ID = '12345'; +const RUN_URL = `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${RUN_ID}`; +const NAME_A = 'dsr1_conc32_b200'; +const NAME_B = 'dsr1_conc64_b200'; +const source = (name: string) => `power_validation_${name}.json`; +const csvName = (name: string) => `gpu_metrics_${name}`; +const SOURCE_A = source(NAME_A); +const SOURCE_B = source(NAME_B); +const START = Date.parse('2026-03-01T00:00:00Z') / 1000; +const workflowRun = { + id: 1, + githubRunId: Number(RUN_ID), + runAttempt: 1, + name: 'Run Sweep', + date: '2026-03-01', + htmlUrl: RUN_URL, + headBranch: 'main', + headSha: 'abc123', + conclusion: 'success', + status: 'completed', + createdAt: '2026-03-01T00:00:00Z', +}; + +function csv(power: number): string { + return [ + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]', + `2026/03/01 00:00:00.000, 0, ${power} W, 65, 1500 MHz, 2000 MHz, 95 %, 80 %`, + `2026/03/01 00:00:01.000, 0, ${power + 10} W, 65, 1500 MHz, 2000 MHz, 95 %, 80 %`, + ].join('\n'); +} + +function stored(name: string, power = 100): GpuMetricSeries { + return { + id: 1, + artifactName: csvName(name), + configKey: name, + fileName: 'gpu_metrics.csv', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 2, + gpuCount: 1, + startedAt: '2026-03-01T00:00:00Z', + endedAt: '2026-03-01T00:00:01Z', + sidecars: { + seriesInventory: [{ fileName: 'gpu_metrics.csv', sampleCount: 2 }], + }, + benchmarkResultIds: [1], + stats: [], + data: parseCsvData(csv(power)).map((row, index) => ({ + ...row, + timestamp: new Date((START + index) * 1000).toISOString(), + })), + }; +} + +function request(body: unknown, prefix = 'dsr1_', raw = false): NextRequest { + const params = new URLSearchParams({ + runId: RUN_ID, + series: 'power', + prefix, + }); + return new NextRequest(`http://localhost/api/gpu-metrics?${params}`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: raw ? String(body) : JSON.stringify(body), + }); +} + +beforeEach(() => { + readRun.mockReset().mockResolvedValue({ workflowRun, series: [stored(NAME_A)] }); + vi.stubEnv('DATABASE_READONLY_URL', 'postgresql://readonly.example.test/test'); + vi.stubEnv('GITHUB_TOKEN', 'test-token'); + vi.stubGlobal('fetch', vi.fn().mockRejectedValue(new Error('Unexpected GitHub request'))); +}); + +afterEach(() => { + vi.unstubAllEnvs(); + vi.unstubAllGlobals(); +}); + +describe('Timeline requested-source coverage', () => { + it('keeps known missing hosts as 503 even when another requested series is healthy', async () => { + const partial = stored(NAME_B, 300); + partial.sidecars.seriesInventory = [ + { fileName: 'gpu_metrics.csv', sampleCount: 2 }, + { fileName: 'host-b/gpu_metrics.csv', sampleCount: 2 }, + ]; + readRun.mockResolvedValue({ + workflowRun, + series: [stored(NAME_A), partial], + }); + vi.stubEnv('GITHUB_TOKEN', ''); + const response = await POST(request({ sources: [SOURCE_A, SOURCE_B] })); + expect(readRun).toHaveBeenCalledWith({}, Number(RUN_ID), { + prefix: 'dsr1_', + sourceResults: [NAME_A, NAME_B], + }); + expect(response.status).toBe(503); + expect(await response.json()).toMatchObject({ + code: 'STORED_TELEMETRY_INCOMPLETE', + artifact: csvName(NAME_B), + }); + }); +}); diff --git a/packages/app/src/app/api/gpu-metrics/route.test.ts b/packages/app/src/app/api/gpu-metrics/route.test.ts index c34a2147f..2b1dd8f15 100644 --- a/packages/app/src/app/api/gpu-metrics/route.test.ts +++ b/packages/app/src/app/api/gpu-metrics/route.test.ts @@ -1,22 +1,37 @@ import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest'; +import type * as GpuPowerTypes from '@/components/gpu-power/types'; -const { mockParseCsvData } = vi.hoisted(() => ({ - mockParseCsvData: vi.fn((csv: string) => { - if (csv.trim().length === 0) return []; - return [ - { - timestamp: '2026-03-01T00:00:00Z', - index: 0, - power: 300, - temperature: 65, - smClock: 1500, - memClock: 2000, - gpuUtil: 95, - memUtil: 80, - }, - ]; - }), -})); +const { mockParseCsvData, zipArchives } = vi.hoisted(() => { + interface ZipEntry { + entryName: string; + data: string; + } + const csvArchive: ZipEntry[] = [ + { entryName: 'gpu_metrics_0.csv', data: 'timestamp,index,power\n2026-03-01T00:00:00Z,0,300' }, + ]; + return { + mockParseCsvData: vi.fn((csv: string): GpuPowerTypes.GpuMetricRow[] => { + if (csv.trim().length === 0) return []; + return [ + { + timestamp: '2026-03-01T00:00:00Z', + index: 0, + power: 300, + temperature: 65, + smClock: 1500, + memClock: 2000, + gpuUtil: 95, + memUtil: 80, + }, + ]; + }), + /** + * Entries the adm-zip mock serves, keyed by the downloaded buffer's text. + * An empty download (the default in older tests) is the one-CSV artifact. + */ + zipArchives: { byKey: new Map(), csvArchive }, + }; +}); vi.mock('@semianalysisai/inferencex-constants', () => ({ GITHUB_API_BASE: 'https://api.github.com', @@ -28,36 +43,60 @@ vi.mock('@/components/gpu-power/types', () => ({ parseCsvData: mockParseCsvData, })); +const { mockGetGpuMetricsForRun } = vi.hoisted(() => ({ + mockGetGpuMetricsForRun: vi.fn(), +})); + +vi.mock('@semianalysisai/inferencex-db/connection', () => ({ + getDb: () => ({}), +})); + +vi.mock('@semianalysisai/inferencex-db/queries/gpu-metrics', () => ({ + getGpuMetricsForRun: mockGetGpuMetricsForRun, +})); + vi.mock('adm-zip', () => { - const csvContent = 'timestamp,index,power\n2026-03-01T00:00:00Z,0,300'; class MockAdmZip { + private readonly key: string; + constructor(buffer: Buffer) { + this.key = buffer.toString('utf8'); + } getEntries() { - return [ - { - entryName: 'gpu_metrics_0.csv', - isDirectory: false, - getData: () => Buffer.from(csvContent), - }, - ]; + const entries = zipArchives.byKey.get(this.key) ?? zipArchives.csvArchive; + return entries.map((entry) => ({ + entryName: entry.entryName, + isDirectory: false, + getData: () => Buffer.from(entry.data), + })); } } return { default: MockAdmZip }; }); -import { GET } from './route'; +import { databasePayloadToResponse, GET, readGpuMetricsForView } from './route'; import { NextRequest } from 'next/server'; const originalFetch = globalThis.fetch; let origToken: string | undefined; +let origReadonlyUrl: string | undefined; function req(url: string): NextRequest { return new NextRequest(new URL(url, 'http://localhost')); } +function nvidiaRow(second: number, power: number): string { + return `2026/03/01 00:00:0${second}.000, 0, ${power} W, 65, 1500 MHz, 2000 MHz, 95 %, 80 %`; +} + beforeEach(() => { vi.clearAllMocks(); + zipArchives.byKey.clear(); origToken = process.env.GITHUB_TOKEN; + origReadonlyUrl = process.env.DATABASE_READONLY_URL; process.env.GITHUB_TOKEN = 'test-gh-token'; + // No readonly URL: the GitHub fallback is exercised unless a test opts in. + delete process.env.DATABASE_READONLY_URL; + mockGetGpuMetricsForRun.mockResolvedValue(null); }); afterEach(() => { @@ -67,6 +106,258 @@ afterEach(() => { } else { process.env.GITHUB_TOKEN = origToken; } + if (origReadonlyUrl === undefined) { + delete process.env.DATABASE_READONLY_URL; + } else { + process.env.DATABASE_READONLY_URL = origReadonlyUrl; + } +}); + +const storedRunPayload = { + workflowRun: { + id: 7, + githubRunId: 34557177019, + runAttempt: 1, + name: 'Run Sweep - dsr1 fp4 b200', + date: '2026-09-11', + htmlUrl: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34557177019', + headBranch: 'main', + headSha: 'deadbeef', + conclusion: 'success', + status: 'completed', + createdAt: '2026-09-11T04:00:00.000Z', + }, + series: [ + { + id: 1, + artifactName: 'gpu_metrics_dsr1_conc32_b200-x_0', + configKey: 'dsr1_conc32_b200-x_0', + fileName: 'gpu_metrics.csv', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 2, + gpuCount: 1, + startedAt: '2026-09-11T04:19:41.982Z', + endedAt: '2026-09-11T04:19:42.990Z', + sidecars: {}, + benchmarkResultIds: [10], + stats: [], + data: [ + { + timestamp: '2026-09-11T04:19:41.982Z', + index: 0, + power: 187.8, + temperature: 33, + smClock: 120, + memClock: 3996, + gpuUtil: 0, + memUtil: 0, + }, + { + timestamp: '2026-09-11T04:19:42.990Z', + index: 0, + power: 912.1, + temperature: 61, + smClock: 1965, + memClock: 3996, + gpuUtil: 98, + memUtil: 74, + }, + ], + }, + { + id: 2, + artifactName: 'gpu_metrics_multinode_b200-x_0', + configKey: 'multinode_b200-x_0', + fileName: 'results/gpu_metrics_rank0.csv', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 0, + gpuCount: 0, + startedAt: '2026-09-11T04:19:41.982Z', + endedAt: '2026-09-11T04:19:41.982Z', + sidecars: {}, + benchmarkResultIds: [11], + stats: [], + data: [], + }, + { + id: 3, + artifactName: 'gpu_metrics_multinode_b200-x_0', + configKey: 'multinode_b200-x_0', + fileName: 'results/gpu_metrics_rank1.csv', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 0, + gpuCount: 0, + startedAt: '2026-09-11T04:19:41.982Z', + endedAt: '2026-09-11T04:19:41.982Z', + sidecars: {}, + benchmarkResultIds: [11], + stats: [], + data: [], + }, + ], +}; + +describe('databasePayloadToResponse', () => { + it('shapes the stored digest like the GitHub payload and disambiguates multinode CSVs', () => { + const response = databasePayloadToResponse(storedRunPayload); + expect(response.source).toBe('database'); + expect(response.runInfo).toEqual({ + id: 34557177019, + name: 'Run Sweep - dsr1 fp4 b200', + branch: 'main', + sha: 'deadbeef', + createdAt: '2026-09-11T04:00:00.000Z', + url: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34557177019', + conclusion: 'success', + status: 'completed', + }); + expect(response.artifacts.map((artifact) => artifact.name)).toEqual([ + 'gpu_metrics_dsr1_conc32_b200-x_0', + 'gpu_metrics_multinode_b200-x_0/results/gpu_metrics_rank0.csv', + 'gpu_metrics_multinode_b200-x_0/results/gpu_metrics_rank1.csv', + ]); + expect(response.artifacts[0]!.data).toHaveLength(2); + expect(response.artifacts[0]!.series?.benchmarkResultIds).toEqual([10]); + expect(response.artifacts[0]!.series).not.toHaveProperty('data'); + }); +}); + +describe('GET /api/gpu-metrics — database first', () => { + it('serves the stored digest without touching GitHub when the run is ingested', async () => { + process.env.DATABASE_READONLY_URL = 'postgresql://readonly.example.test/db'; + mockGetGpuMetricsForRun.mockResolvedValueOnce(storedRunPayload); + globalThis.fetch = vi.fn(); + + const res = await GET(req('/api/gpu-metrics?runId=34557177019')); + expect(res.status).toBe(200); + const body = await res.json(); + expect(body.source).toBe('database'); + expect(body.artifacts).toHaveLength(3); + expect(mockGetGpuMetricsForRun).toHaveBeenCalledWith({}, 34557177019, { + prefix: null, + sourceResults: null, + }); + expect(globalThis.fetch).not.toHaveBeenCalled(); + }); + + it('passes selectors to the reader and preserves exact host labels for a scoped view', async () => { + process.env.DATABASE_READONLY_URL = 'postgresql://readonly.example.test/db'; + const names = databasePayloadToResponse(storedRunPayload).artifacts.map((entry) => entry.name); + mockGetGpuMetricsForRun.mockResolvedValueOnce({ + ...storedRunPayload, + artifactNames: names, + series: [storedRunPayload.series[2]], + }); + globalThis.fetch = vi.fn(); + const response = await readGpuMetricsForView( + req('/api/gpu-metrics?runId=34557177019'), + names[2]!, + ); + expect(await response.json()).toMatchObject({ + artifactNames: names, + artifacts: [{ name: names[2] }], + }); + expect(mockGetGpuMetricsForRun).toHaveBeenCalledWith({}, 34557177019, { + prefix: null, + sourceResults: null, + artifact: names[2], + }); + mockGetGpuMetricsForRun.mockResolvedValueOnce({ + ...storedRunPayload, + artifactNames: names, + series: [], + }); + const missing = await readGpuMetricsForView( + req('/api/gpu-metrics?runId=34557177019'), + 'missing', + ); + expect(await missing.json()).toMatchObject({ artifactNames: names, artifacts: [] }); + expect(globalThis.fetch).not.toHaveBeenCalled(); + }); + + it('reports a database failure separately instead of treating it as missing data', async () => { + process.env.DATABASE_READONLY_URL = 'postgresql://readonly.example.test/db'; + mockGetGpuMetricsForRun.mockRejectedValueOnce(new Error('relation does not exist')); + globalThis.fetch = vi.fn().mockResolvedValueOnce({ ok: false, status: 404 }); + + const res = await GET(req('/api/gpu-metrics?runId=99')); + expect(res.status).toBe(503); + expect(await res.json()).toMatchObject({ code: 'DATABASE_UNAVAILABLE' }); + expect(globalThis.fetch).not.toHaveBeenCalled(); + }); + + it('checks retained file/sample coverage before using a truncated host CSV fallback', async () => { + process.env.DATABASE_READONLY_URL = 'postgresql://readonly.example.test/db'; + const { parseCsvData } = await vi.importActual( + '@/components/gpu-power/types', + ); + const header = + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]'; + const hostA = [header, nvidiaRow(0, 300), nvidiaRow(0, 999), nvidiaRow(1, 310)].join('\n'); + // host-b lost its GPU 1 row: one of the two retained samples. + const hostB = [header, nvidiaRow(0, 500)].join('\n'); + zipArchives.byKey.set('coverage', [ + { entryName: 'host-a/gpu_metrics.csv', data: hostA }, + { entryName: 'host-b/gpu_metrics.csv', data: hostB }, + ]); + mockParseCsvData.mockImplementationOnce(parseCsvData).mockImplementationOnce(parseCsvData); + mockGetGpuMetricsForRun.mockResolvedValueOnce({ + ...storedRunPayload, + series: [ + { + ...storedRunPayload.series[0], + artifactName: 'gpu_metrics_live', + fileName: 'host-a/gpu_metrics.csv', + sidecars: { + seriesInventory: [ + { fileName: 'host-a/gpu_metrics.csv', sampleCount: 2 }, + { fileName: 'host-b/gpu_metrics.csv', sampleCount: 2 }, + ], + }, + }, + ], + }); + globalThis.fetch = vi + .fn() + .mockResolvedValueOnce({ + ok: true, + json: () => + Promise.resolve({ + id: 34557177019, + name: 'Sweep', + head_branch: 'main', + head_sha: 'abc', + created_at: '2026-03-01T00:00:00Z', + html_url: 'https://example.test/run', + status: 'completed', + conclusion: 'success', + }), + }) + .mockResolvedValueOnce({ + ok: true, + json: () => + Promise.resolve({ + artifacts: [ + { id: 1, name: 'gpu_metrics_live', archive_download_url: 'https://example.test/zip' }, + ], + }), + }) + .mockResolvedValueOnce({ + ok: true, + headers: new Headers(), + arrayBuffer: () => Promise.resolve(new TextEncoder().encode('coverage').buffer), + }); + const res = await GET(req('/api/gpu-metrics?runId=34557177019&series=power')); + expect(res.status).toBe(503); + expect(await res.json()).toMatchObject({ + code: 'STORED_TELEMETRY_INCOMPLETE', + artifact: 'gpu_metrics_live', + }); + expect(globalThis.fetch).toHaveBeenCalledTimes(3); + }); }); describe('GET /api/gpu-metrics', () => { diff --git a/packages/app/src/app/api/gpu-metrics/route.ts b/packages/app/src/app/api/gpu-metrics/route.ts index 59a607f35..b9f351d66 100644 --- a/packages/app/src/app/api/gpu-metrics/route.ts +++ b/packages/app/src/app/api/gpu-metrics/route.ts @@ -1,10 +1,70 @@ /** - * DO NOT ADD CACHING (blob, CDN, or unstable_cache) to this route. - * It fetches live GitHub Actions artifacts which change while a run is in progress. + * PowerX telemetry for one GitHub Actions run. + * + * Reads the ingest-time telemetry digest first (migration 016: series, + * samples, per-GPU statistics, point links). Runs that have not been ingested + * yet, including runs still in progress, fall back to the live GitHub + * artifacts exactly as before. + * + * DO NOT ADD CACHING (blob, CDN, or unstable_cache) to this route. The + * fallback fetches live GitHub Actions artifacts which change while a run is + * in progress, and stored telemetry must reflect a successful re-ingest immediately. + * + * Two telemetry collectors publish GPU power for a run: + * - single-node runners (nvidia-smi / amd-smi) publish one `gpu_metrics_` + * CSV artifact per benchmark config; + * - Slurm / Dynamo disaggregated runners (DCGM) publish one `power_audit_` + * bundle per concurrency sweep, holding the sweep's samples plus one + * `power_validation_*.json` window per config + * (`components/gpu-power/power-audit-bundle.ts`). + * + * Two response shapes: + * - default: every `gpu_metrics_*` artifact's parsed rows (the `/gpu-metrics` + * page), from the stored digest when the run is ingested, else from GitHub; + * bundles are ignored on the GitHub path; + * - `series=power`: compact per-GPU watt series bucketed to one second + * (`components/gpu-power/power-series.ts`) for the PowerX timeline, from + * CSV artifacts and from bundles cut per validation window. The timeline + * joins them to chart points by `source` (bundle) or artifact name (CSV). + * Persisted samples and validation windows use the same bucketing/cut + * transform as artifacts, so historical runs survive artifact expiry. + * `prefix=` narrows either shape to the artifacts of + * one model / workload / precision so a full nightly sweep is not downloaded + * for one chart. A bundle names a whole sweep, so it also matches when the + * prefix extends past its name into the per-concurrency suffix. + * Timeline POSTs sorted validation basenames in `{ sources: [...] }` to recover + * missing siblings while keeping fully covered DB reads independent of GitHub. + * `sourceCoverage` describes those requested identities, not full-run completeness. */ import { type NextRequest, NextResponse } from 'next/server'; -import { parseCsvData } from '@/components/gpu-power/types'; +import { getDb } from '@semianalysisai/inferencex-db/connection'; +import { + getGpuMetricsForRun, + type GpuMetricsRunPayload, + type GpuMetricsRunSelection, +} from '@semianalysisai/inferencex-db/queries/gpu-metrics'; + +import type { + GpuMetricRow, + GpuPowerRunInfo, + GpuMetricsArtifact, + GpuPowerApiResponse, +} from '@/components/gpu-power/types'; +import { + cutPowerAuditBundle, + isPowerAuditBundleEntry, + parsePowerCsvData, +} from '@/components/gpu-power/power-audit-bundle'; +import { + bucketPowerFiles, + parseTelemetryTimestampUtc, + type GpuPowerSeries, +} from '@/components/gpu-power/power-series'; +import { + storedPowerSeries, + StoredTelemetryIncompleteError, +} from '@/components/gpu-power/stored-power-series'; import { downloadGithubArtifact, extractZipEntries, @@ -12,12 +72,270 @@ import { fetchGithubWorkflowRun, getGithubToken, normalizeGithubRunInfo, + readZipEntries, + type GithubArtifact, type GithubWorkflowRun, } from '@/lib/github-artifacts'; const MAX_ARTIFACT_BYTES = 50 * 1024 * 1024; +/** Bundles carry a whole sweep (215 MB seen for nw8); only the power entries are decoded. */ +const MAX_BUNDLE_BYTES = 256 * 1024 * 1024; +/** Bundle downloads are latency-bound; match the other artifact routes' budget. */ +export const maxDuration = 300; +/** Parallel artifact downloads; GitHub's zip redirects are latency-bound, not CPU-bound. */ +const DOWNLOAD_CONCURRENCY = 4; +const ARTIFACT_PREFIX = 'gpu_metrics_'; +const BUNDLE_PREFIX = 'power_audit_'; +/** RESULT_FILENAME characters: model, workload, precision, framework, parallelism, host, hash. */ +const PREFIX_PATTERN = /^[A-Za-z0-9._-]{1,200}$/u; +const SOURCE_PATTERN = /^power_validation_[A-Za-z0-9._-]{1,200}\.json$/u; +const MAX_REQUEST_BYTES = 256 * 1024; + +export type GpuMetricsSource = 'database' | 'github'; + +export type GpuMetricsArtifactPayload = GpuMetricsArtifact; + +export interface GpuMetricsRouteResponse extends GpuPowerApiResponse { + source: GpuMetricsSource; + artifactNames?: string[]; +} + +/** Shape the stored digest like the GitHub payload so the explorer is source-agnostic. */ +export function databasePayloadToResponse(payload: GpuMetricsRunPayload): GpuMetricsRouteResponse { + const run = payload.workflowRun; + const filesPerArtifact = new Map(); + for (const series of payload.series) { + filesPerArtifact.set(series.artifactName, (filesPerArtifact.get(series.artifactName) ?? 0) + 1); + } + return { + source: 'database', + ...(payload.artifactNames ? { artifactNames: payload.artifactNames } : {}), + runInfo: { + id: run.githubRunId, + name: run.name, + branch: run.headBranch ?? '', + sha: run.headSha ?? '', + createdAt: run.createdAt ?? `${run.date}T00:00:00Z`, + url: + run.htmlUrl ?? + `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${run.githubRunId}`, + conclusion: run.conclusion ?? '', + status: run.status ?? '', + }, + artifacts: payload.series.map(({ data, ...series }) => ({ + // Multinode uploads carry one CSV per node; keep them distinguishable. + name: + (filesPerArtifact.get(series.artifactName) ?? 1) > 1 || + payload.artifactNames?.includes(`${series.artifactName}/${series.fileName}`) + ? `${series.artifactName}/${series.fileName}` + : series.artifactName, + data, + series, + })), + }; +} + +/** The GitHub path also carries the bundle series the timeline draws. */ +interface GithubArtifactPayload { + name: string; + files: { name: string; data: GpuMetricRow[] }[]; +} + +interface GithubGpuMetricsResponse extends Omit { + artifacts: GithubArtifactPayload[]; + source: 'github'; + bundleSeries: GpuPowerSeries[]; +} + +type TelemetryJob = + | { kind: 'csv'; artifact: GithubArtifact } + | { kind: 'bundle'; artifact: GithubArtifact }; + +type TelemetryResult = + | { kind: 'csv'; parsed: GithubArtifactPayload } + | { kind: 'bundle'; series: GpuPowerSeries[] }; + +/** Fetches one artifact zip, or `null` (with a warning) when it fails or exceeds `maxBytes`. */ +async function downloadZip( + artifact: GithubArtifact, + githubToken: string, + maxBytes: number, +): Promise { + const dlResp = await downloadGithubArtifact(artifact.archive_download_url, githubToken); + if (!dlResp.ok) { + console.warn(`Failed to download artifact ${artifact.name}: ${dlResp.statusText}`); + return null; + } + + const contentLength = dlResp.headers.get('Content-Length'); + if (contentLength && parseInt(contentLength, 10) > maxBytes) { + console.warn(`Artifact ${artifact.name} exceeds ${maxBytes / (1024 * 1024)} MB, skipping`); + return null; + } + return Buffer.from(await dlResp.arrayBuffer()); +} + +async function downloadArtifact( + artifact: GithubArtifact, + githubToken: string, +): Promise { + const buffer = await downloadZip(artifact, githubToken, MAX_ARTIFACT_BYTES); + if (!buffer) return null; + const contexts = new Map>(); + const contextFiles = extractZipEntries(buffer, '.json', (name, contents) => { + const base = name.slice(name.lastIndexOf('/') + 1); + if (!base.includes('gpu_metrics') || !base.toLowerCase().endsWith('_context.json')) return []; + const context: unknown = JSON.parse(contents); + return context && typeof context === 'object' && !Array.isArray(context) + ? [{ name, context: context as Record }] + : []; + }); + // Ingest uses code-unit filename order, not archive order or the host locale. + contextFiles.sort((a, b) => (a.name < b.name ? -1 : a.name > b.name ? 1 : 0)); + for (const { name, context } of contextFiles) { + const directory = name.slice(0, name.lastIndexOf('/') + 1); + if (!contexts.has(directory)) contexts.set(directory, context); + } + const files = extractZipEntries( + buffer, + '.csv', + (entryName, contents) => { + const directory = entryName.slice(0, entryName.lastIndexOf('/') + 1); + const data = parsePowerCsvData(contents, contexts.get(directory) ?? null); + return data.length > 0 ? [{ name: entryName, data }] : []; + }, + (entryName, error) => { + console.warn(`Failed to parse CSV ${entryName} from ${artifact.name}:`, error); + }, + ); + return files.length > 0 ? { name: artifact.name, files } : null; +} + +async function downloadBundle( + artifact: GithubArtifact, + githubToken: string, +): Promise { + const buffer = await downloadZip(artifact, githubToken, MAX_BUNDLE_BYTES); + if (!buffer) return null; + const series = cutPowerAuditBundle( + artifact.name, + readZipEntries(buffer, isPowerAuditBundleEntry), + ); + return series.length > 0 ? series : null; +} + +/** + * One artifact download and parse. A corrupt or truncated archive is that + * artifact's failure alone: it is logged and skipped so the other series of + * the run still reach the chart (the client would otherwise retry the whole + * multi-hundred-megabyte request). + */ +async function runJob(job: TelemetryJob, githubToken: string): Promise { + try { + if (job.kind === 'csv') { + const parsed = await downloadArtifact(job.artifact, githubToken); + return parsed ? { kind: 'csv', parsed } : null; + } + const series = await downloadBundle(job.artifact, githubToken); + return series ? { kind: 'bundle', series } : null; + } catch (error) { + console.warn(`Failed to read artifact ${job.artifact.name}:`, error); + return null; + } +} + +/** Downloads in listing order with a bounded number of requests in flight. */ +async function downloadTelemetry( + jobs: TelemetryJob[], + githubToken: string, +): Promise { + const results: (TelemetryResult | null)[] = Array.from({ length: jobs.length }, () => null); + let next = 0; + const worker = async () => { + while (next < jobs.length) { + const index = next++; + results[index] = await runJob(jobs[index], githubToken); + } + }; + await Promise.all(Array.from({ length: Math.min(DOWNLOAD_CONCURRENCY, jobs.length) }, worker)); + return results.filter((result): result is TelemetryResult => result !== null); +} + +/** + * A bundle names a whole sweep, while the client's prefix (the longest common + * prefix of its points' validation names) may run into the `_sa-bench_…_conc` + * suffix; either side being a prefix of the other selects the bundle. + */ +function isWantedBundle(name: string, prefix: string | null): boolean { + if (!name.startsWith(BUNDLE_PREFIX)) return false; + if (prefix === null) return true; + const wanted = `${BUNDLE_PREFIX}${prefix}`; + return name.startsWith(wanted) || wanted.startsWith(name); +} + +/** Validation basenames identify individual windows, including siblings in one bundle. */ +function seriesSource(series: GpuPowerSeries): string | null { + return ( + series.source ?? + (series.artifact.startsWith(ARTIFACT_PREFIX) + ? `power_validation_${series.artifact.slice(ARTIFACT_PREFIX.length)}.json` + : null) + ); +} + +function isRequestedArtifact(name: string, sources: string[] | null): boolean { + return ( + sources === null || + sources.some((source) => { + const result = source.slice('power_validation_'.length, -'.json'.length); + return name === `${ARTIFACT_PREFIX}${result}` || isWantedBundle(name, result); + }) + ); +} + +function sourceCoverage(series: GpuPowerSeries[], sources: string[] | null) { + const available = new Set(series.map(seriesSource)); + const missingSources = sources?.filter((source) => !available.has(source)) ?? []; + return { + status: sources === null ? 'unknown' : missingSources.length > 0 ? 'incomplete' : 'complete', + missingSources, + }; +} + +function powerSeriesResponse( + source: GpuMetricsSource, + runInfo: GpuPowerRunInfo, + series: GpuPowerSeries[], + sources: string[] | null, +) { + return NextResponse.json( + { source, runInfo, series, sourceCoverage: sourceCoverage(series, sources) }, + { headers: { 'Cache-Control': 'no-store' } }, + ); +} -async function fetchGpuMetrics(runId: string) { +/** + * Narrows a stored run to the artifacts a `prefix` names, mirroring the GitHub + * listing filter so both sources answer the same request the same way. + */ +function filterArtifactsByPrefix( + artifacts: GpuMetricsArtifactPayload[], + prefix: string | null, +): GpuMetricsArtifactPayload[] { + if (prefix === null) return artifacts; + const wanted = `${ARTIFACT_PREFIX}${prefix}`; + return artifacts.filter((artifact) => { + const name = artifact.series?.artifactName ?? artifact.name; + return name.startsWith(wanted) || isWantedBundle(name, prefix); + }); +} + +async function fetchGpuMetricsFromGithub( + runId: string, + prefix: string | null, + includeBundles: boolean, + sources: string[] | null = null, +): Promise { const githubToken = getGithubToken(); if (!githubToken) throw new Error('GitHub token not configured'); @@ -27,57 +345,292 @@ async function fetchGpuMetrics(runId: string) { const artifacts = await fetchGithubRunArtifacts(runId, githubToken); - const gpuArtifacts = artifacts.filter((a) => a.name.startsWith('gpu_metrics')); - if (gpuArtifacts.length === 0) throw new Error('No gpu_metrics artifacts found for this run'); - - const parsedArtifacts: { name: string; data: ReturnType }[] = []; - for (const artifact of gpuArtifacts) { - const dlResp = await downloadGithubArtifact(artifact.archive_download_url, githubToken); - if (!dlResp.ok) { - console.warn(`Failed to download artifact ${artifact.name}: ${dlResp.statusText}`); - continue; + // `eval_gpu_metrics_*` artifacts are excluded by the bare `gpu_metrics` test. + const wanted = prefix ? `${ARTIFACT_PREFIX}${prefix}` : 'gpu_metrics'; + const jobs: TelemetryJob[] = artifacts + .filter((a) => a.name.startsWith(wanted) && isRequestedArtifact(a.name, sources)) + .map((artifact) => ({ kind: 'csv', artifact })); + if (includeBundles) { + for (const artifact of artifacts) { + if (isWantedBundle(artifact.name, prefix) && isRequestedArtifact(artifact.name, sources)) + jobs.push({ kind: 'bundle', artifact }); } - - const contentLength = dlResp.headers.get('Content-Length'); - if (contentLength && parseInt(contentLength, 10) > MAX_ARTIFACT_BYTES) { - console.warn(`Artifact ${artifact.name} exceeds 50 MB, skipping`); - continue; - } - - const rows = extractZipEntries( - Buffer.from(await dlResp.arrayBuffer()), - '.csv', - (_entryName, contents) => parseCsvData(contents), - (entryName, error) => { - console.warn(`Failed to parse CSV ${entryName} from ${artifact.name}:`, error); - }, + } + if (jobs.length === 0) { + throw new Error( + includeBundles + ? 'No telemetry artifacts (gpu_metrics or power_audit) found for this run' + : 'No gpu_metrics artifacts found for this run', ); - if (rows.length > 0) parsedArtifacts.push({ name: artifact.name, data: rows }); } - if (parsedArtifacts.length === 0) throw new Error('No Chip metrics data found in artifacts'); + const results = await downloadTelemetry(jobs, githubToken); + if (results.length === 0) throw new Error('No Chip metrics data found in artifacts'); + const parsedArtifacts: GithubArtifactPayload[] = []; + const bundleSeries: GpuPowerSeries[] = []; + for (const result of results) { + if (result.kind === 'csv') parsedArtifacts.push(result.parsed); + else bundleSeries.push(...result.series); + } return { - runInfo: normalizeGithubRunInfo(run), + source: 'github', + runInfo: normalizeGithubRunInfo(run) as GpuPowerRunInfo, artifacts: parsedArtifacts, + bundleSeries, }; } -export async function GET(request: NextRequest) { - const runId = request.nextUrl.searchParams.get('runId'); +async function fetchGpuMetricsFromDatabase( + runId: string, + selection: GpuMetricsRunSelection, +): Promise { + if (!process.env.DATABASE_READONLY_URL) return null; + const payload = await getGpuMetricsForRun(getDb(), Number(runId), selection); + return payload ? databasePayloadToResponse(payload) : null; +} + +export function GET(request: NextRequest) { + return readGpuMetrics(request, null); +} + +/** Selecting a host must not discard the explorer's sibling artifact choices. */ +export function readGpuMetricsForView(request: NextRequest, artifact: string | null) { + return readGpuMetrics(request, null, artifact); +} + +/** Read-only Timeline transport; the body avoids URL limits for a run's point identities. */ +export async function POST(request: NextRequest) { + const reader = request.body?.getReader(); + const chunks: Uint8Array[] = []; + let bytes = 0; + try { + if (reader) { + while (true) { + const { done, value } = await reader.read(); + if (done) break; + bytes += value.byteLength; + if (bytes > MAX_REQUEST_BYTES) { + await reader.cancel(); + return NextResponse.json({ error: 'Request body exceeds 256 KiB' }, { status: 413 }); + } + chunks.push(value); + } + } + const body: unknown = JSON.parse(Buffer.concat(chunks).toString('utf8')); + const sources = + body && typeof body === 'object' && !Array.isArray(body) && 'sources' in body + ? body.sources + : null; + if ( + !Array.isArray(sources) || + sources.length === 0 || + sources.length > 1000 || + !sources.every( + (source): source is string => typeof source === 'string' && SOURCE_PATTERN.test(source), + ) + ) { + return NextResponse.json( + { error: 'sources must contain 1–1000 power_validation_.json basenames' }, + { status: 400 }, + ); + } + if (request.nextUrl.searchParams.get('series') !== 'power') { + return NextResponse.json({ error: 'POST requires series=power' }, { status: 400 }); + } + const prefix = request.nextUrl.searchParams.get('prefix'); + if ( + prefix !== null && + sources.some((source) => !source.startsWith(`power_validation_${prefix}`)) + ) { + return NextResponse.json({ error: 'Every source must match prefix' }, { status: 400 }); + } + return readGpuMetrics(request, [...new Set(sources)].sort()); + } catch { + return NextResponse.json({ error: 'Request body must be valid JSON' }, { status: 400 }); + } finally { + reader?.releaseLock(); + } +} + +async function readGpuMetrics( + request: NextRequest, + sources: string[] | null, + selectedArtifact?: string | null, +) { + const params = request.nextUrl.searchParams; + const runId = params.get('runId'); if (!runId || !/^\d+$/u.test(runId)) { return NextResponse.json({ error: 'runId must be a numeric workflow run ID' }, { status: 400 }); } + const prefix = params.get('prefix'); + if (prefix !== null && !PREFIX_PATTERN.test(prefix)) { + return NextResponse.json( + { error: 'prefix must be a RESULT_FILENAME prefix (letters, digits, . _ -)' }, + { status: 400 }, + ); + } + const series = params.get('series'); + if (series !== null && series !== 'power') { + return NextResponse.json({ error: 'series must be "power" when present' }, { status: 400 }); + } + + let stored: GpuMetricsRouteResponse | null; + try { + stored = await fetchGpuMetricsFromDatabase(runId, { + prefix, + sourceResults: + sources?.map((source) => source.slice('power_validation_'.length, -'.json'.length)) ?? null, + ...(selectedArtifact === undefined ? {} : { artifact: selectedArtifact }), + }); + } catch (error) { + // Missing data may use live artifacts; a failed read cannot establish absence. + console.error(`gpu-metrics: database lookup failed for run ${runId}:`, error); + return NextResponse.json( + { + error: 'Stored telemetry is temporarily unavailable. Retry the request.', + code: 'DATABASE_UNAVAILABLE', + }, + { status: 503, headers: { 'Cache-Control': 'no-store' } }, + ); + } + const incomplete: StoredTelemetryIncompleteError[] = []; + const databaseSeries: GpuPowerSeries[] = []; + let githubFallbackStarted = false; try { - const data = await fetchGpuMetrics(runId); - return NextResponse.json(data); + if (stored) { + const artifacts = filterArtifactsByPrefix(stored.artifacts, prefix).filter((artifact) => + isRequestedArtifact(artifact.series?.artifactName ?? artifact.name, sources), + ); + if (series === 'power') { + const groups = Map.groupBy( + artifacts.flatMap((artifact) => + artifact.series ? [{ ...artifact.series, data: artifact.data }] : [], + ), + (entry) => entry.artifactName, + ); + for (const entries of groups.values()) { + try { + databaseSeries.push( + ...storedPowerSeries(entries).filter( + (entry) => sources === null || sources.includes(seriesSource(entry) ?? ''), + ), + ); + } catch (error) { + if (!(error instanceof StoredTelemetryIncompleteError)) throw error; + incomplete.push(error); + console.warn( + 'gpu-metrics: incomplete stored telemetry, trying artifacts:', + error.message, + ); + } + } + if ( + incomplete.length === 0 && + databaseSeries.length > 0 && + sourceCoverage(databaseSeries, sources).missingSources.length === 0 + ) { + return powerSeriesResponse('database', stored.runInfo, databaseSeries, sources); + } + } else if (artifacts.length > 0 || stored.artifactNames) { + return NextResponse.json( + { ...stored, artifacts }, + { headers: { 'Cache-Control': 'no-store' } }, + ); + } + } + if (series === 'power') { + githubFallbackStarted = true; + const { runInfo, artifacts, bundleSeries } = await fetchGpuMetricsFromGithub( + runId, + prefix, + true, + sources === null ? null : sourceCoverage(databaseSeries, sources).missingSources, + ); + const powerSeries = artifacts + .map((artifact) => bucketPowerFiles(artifact.name, artifact.files)) + .filter((entry): entry is GpuPowerSeries => entry !== null); + // A fallback response can combine durable history with live recovery. + // Keep healthy stored windows even when their GitHub copies expired, + // and never replace or duplicate them with a live copy. source='github' + // records that this response required fallback, not that every row is live. + const databaseSources = new Set(databaseSeries.map(seriesSource)); + const combined = [ + ...databaseSeries, + ...[...powerSeries, ...bundleSeries].filter( + (entry) => + !databaseSources.has(seriesSource(entry)) && + (sources === null || sources.includes(seriesSource(entry) ?? '')), + ), + ]; + for (const missing of incomplete) { + const incompleteArtifact = missing.artifact; + const live = artifacts.find((artifact) => artifact.name === incompleteArtifact); + const inventory = stored?.artifacts.find( + (artifact) => artifact.series?.artifactName === incompleteArtifact, + )?.series?.sidecars.seriesInventory; + // A matching name alone cannot prove that missing hosts/samples recovered. + // Bundle cuts do not retain the raw inventory, so known-incomplete bundles + // require re-ingest; ordinary un-ingested bundle fallback stays available. + if (!live || !Array.isArray(inventory) || inventory.length === 0) throw missing; + const counts = new Map( + live.files.map((file) => [ + file.name, + new Set( + file.data.flatMap((row) => { + const time = parseTelemetryTimestampUtc(row.timestamp); + return time === null || !Number.isInteger(row.index) || !Number.isFinite(row.power) + ? [] + : [`${row.index}:${time}`]; + }), + ).size, + ]), + ); + if ( + !inventory.every( + (expected) => + expected !== null && + typeof expected === 'object' && + typeof expected.fileName === 'string' && + typeof expected.sampleCount === 'number' && + counts.get(expected.fileName) === expected.sampleCount, + ) + ) + throw missing; + } + return powerSeriesResponse('github', runInfo, combined, sources); + } + const live = await fetchGpuMetricsFromGithub(runId, prefix, false); + return NextResponse.json( + { + source: live.source, + runInfo: live.runInfo, + artifacts: live.artifacts.flatMap(({ name, files }) => + files.map((file) => ({ + name: files.length > 1 ? `${name}/${file.name}` : name, + data: file.data, + })), + ), + }, + { headers: { 'Cache-Control': 'no-store' } }, + ); } catch (error) { console.error('Error fetching GPU power data:', error); + const missing = error instanceof StoredTelemetryIncompleteError ? error : incomplete[0]; + if (githubFallbackStarted && !missing && stored && databaseSeries.length > 0) { + return powerSeriesResponse('database', stored.runInfo, databaseSeries, sources); + } return NextResponse.json( - { error: error instanceof Error ? error.message : 'Unknown error occurred' }, - { status: 500 }, + missing + ? { + error: missing.message, + code: 'STORED_TELEMETRY_INCOMPLETE', + artifact: missing.artifact, + } + : { error: error instanceof Error ? error.message : 'Unknown error occurred' }, + { status: missing ? 503 : 500, headers: { 'Cache-Control': 'no-store' } }, ); } } diff --git a/packages/app/src/app/api/v1/gpu-metrics-point/route.test.ts b/packages/app/src/app/api/v1/gpu-metrics-point/route.test.ts new file mode 100644 index 000000000..0fdad8907 --- /dev/null +++ b/packages/app/src/app/api/v1/gpu-metrics-point/route.test.ts @@ -0,0 +1,264 @@ +import { createHash } from 'node:crypto'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { gunzipSync } from 'node:zlib'; +import { PGlite } from '@electric-sql/pglite'; +import { NextRequest } from 'next/server'; +import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest'; +import { ingestGpuMetricsArtifact } from '@semianalysisai/inferencex-db/etl/gpu-metrics-ingest'; +import type * as GpuMetricStatsModule from '@semianalysisai/inferencex-db/lib/gpu-metric-stats'; +import type { GpuMetricSeries } from '@semianalysisai/inferencex-db/queries/gpu-metrics'; +import { getGpuMetricsPointRevision } from '@semianalysisai/inferencex-db/queries/gpu-metrics-revision'; +import { startPowerxBlobFixture } from '../../../../../scripts/powerx-blob-fixture'; + +const algorithm = vi.hoisted(() => ({ version: 1 })); +vi.mock('@semianalysisai/inferencex-db/lib/gpu-metric-stats', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + get GPU_STATS_VERSION() { + return algorithm.version; + }, + }; +}); + +const connection = vi.hoisted(() => ({ getDb: vi.fn() })); +vi.mock('@semianalysisai/inferencex-db/connection', () => connection); +// The query, cache abstraction, Blob SDK, HTTP transport and ETL are real. +import { GET } from './route'; + +let db: PGlite; +type Sql = Parameters[0]; +let sql: Sql; +let blob: Awaited>; +let root: string; +let sampleReads = 0; +const artifactName = 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0'; +const artifact = () => ({ artifactName, artifactDir: root }); + +function client(database: Pick) { + return Object.assign( + async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + if (query.includes('from gpu_metric_samples')) sampleReads++; + const result = await database.query>(query, values); + return result.rows; + }, + { json: JSON.stringify, array: (value: unknown) => value }, + ); +} + +beforeAll(async () => { + db = await PGlite.create(); + for (const name of [ + '001_initial_schema.sql', + '016_gpu_metrics.sql', + '017_gpu_metric_stats_version.sql', + ]) { + await db.exec( + fs.readFileSync(new URL(`../../../../../../db/migrations/${name}`, import.meta.url), 'utf8'), + ); + } + await db.exec(`ALTER TABLE benchmark_results ADD COLUMN power_audit jsonb; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, status, conclusion, created_at, date) + VALUES (1, 34557177019, 1, 'Run Sweep', 'completed', 'success', '2026-09-11', '2026-09-11'); + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, 4, 4, 4, 4); + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, isl, osl, conc, metrics) + VALUES (10, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, '{}'), + (11, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 64, '{}');`); + sql = Object.assign(client(db), { + begin: (fn: (tx: Sql) => Promise) => + db.transaction((tx) => fn(client(tx) as unknown as Sql)), + }) as unknown as Sql; + connection.getDb.mockReturnValue(sql); + blob = await startPowerxBlobFixture(); + for (const [key, value] of Object.entries(blob.env)) vi.stubEnv(key, value); + root = fs.mkdtempSync(path.join(os.tmpdir(), 'powerx-cache-recovery-')); + // Retained run 34175132645, attempt 1; source and excerpt hashes live beside the fixture. + const csv = fs.readFileSync( + new URL('../../../../../../../docs/fixtures/powerx-reingest/nvidia.csv', import.meta.url), + ); + expect(createHash('sha256').update(csv).digest('hex')).toBe( + '833cf864da4618579deeeda89b3ddde6dbf8b7b32b7877fda643f18411ee3481', + ); + fs.writeFileSync(path.join(root, 'gpu_metrics.csv'), csv); +}, 20_000); + +afterAll(async () => { + vi.unstubAllEnvs(); + await blob?.close(); + await db?.close(); + if (root) fs.rmSync(root, { recursive: true, force: true }); +}); + +beforeEach(async () => { + algorithm.version = 1; + await db.exec( + 'TRUNCATE gpu_metric_series RESTART IDENTITY CASCADE; UPDATE benchmark_results SET power_audit = null;', + ); + blob.objects.clear(); + Object.assign(blob.counts, { reads: 0, writes: 0, failedWrites: 0 }); + sampleReads = 0; + // Injected test sidecars: the retained source does not establish its timezone or identity. + fs.writeFileSync( + path.join(root, 'gpu_metrics_context.json'), + JSON.stringify({ timestamp_timezone: 'UTC' }), + ); + fs.writeFileSync( + path.join(root, 'gpu_metrics_identity.csv'), + 'index, uuid, name\n0, GPU-old, NVIDIA B200\n', + ); +}); + +async function request(id: number) { + const response = await GET(new NextRequest(`http://localhost/api/v1/gpu-metrics-point?id=${id}`)); + const bytes = Buffer.from(await response.arrayBuffer()); + const body = + response.headers.get('content-encoding') === 'gzip' + ? gunzipSync(bytes).toString() + : bytes.toString(); + return { response, body: JSON.parse(body) }; +} + +describe('artifact → local DB → real Blob cache → point API recovery', () => { + it('requires live revision checks for successful and missing responses', async () => { + const missing = await request(10); + expect(missing.response.headers.get('cache-control')).toBe('no-store'); + await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: artifact(), + benchmarkResultIds: [10], + }); + const present = await request(10); + expect(present.response.headers.get('cache-control')).toBe('no-store'); + }); + it('refreshes missing points, shared links, timezone and identity repairs without purging', async () => { + const missing = await request(10); + expect(missing.response.status).toBe(404); + await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: artifact(), + benchmarkResultIds: [10], + }); + const old = await request(10); + expect(old.response.status).toBe(200); + const originalSeries: GpuMetricSeries = old.body.series[0]; + expect(originalSeries).toMatchObject({ + sampleCount: 24, + gpuCount: 8, + startedAt: '2026-09-08T07:20:19.279Z', + endedAt: '2026-09-08T07:20:21.279Z', + }); + expect(originalSeries.data).toHaveLength(24); + expect(originalSeries.data.filter((row) => row.index === 0).map((row) => row.power)).toEqual([ + 380.42, 336.61, 337.18, + ]); + expect(blob.objects.size).toBe(1); + const beforeHit = blob.counts.reads; + const beforeSampleReads = sampleReads; + const cached = await request(10); + expect(cached.body).toEqual(old.body); + expect(blob.counts.reads).toBe(beforeHit + 1); + expect(sampleReads).toBe(beforeSampleReads); + + const unlinked = await request(11); + expect(unlinked.response.status).toBe(404); + await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: artifact(), + benchmarkResultIds: [11], + }); + for (const id of [10, 11]) { + const linked = await request(id); + expect(linked.body.series[0].benchmarkResultIds).toEqual([10, 11]); + } + const oldRevision = await getGpuMetricsPointRevision(client(db), 10); + fs.writeFileSync( + path.join(root, 'gpu_metrics_context.json'), + JSON.stringify({ timestamp_timezone: '+08:00' }), + ); + fs.writeFileSync( + path.join(root, 'gpu_metrics_identity.csv'), + 'index, uuid, name\n0, GPU-corrected, NVIDIA B200\n', + ); + await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: artifact(), + benchmarkResultIds: [10, 11], + }); + expect(await getGpuMetricsPointRevision(client(db), 10)).not.toBe(oldRevision); + for (const id of [10, 11]) { + const repaired = await request(id); + expect(repaired.body.series[0].data[0].timestamp).toBe('2026-09-07T23:20:19.279Z'); + expect(repaired.body.series[0].sampleCount).toBe(24); + expect(JSON.stringify(repaired.body.series[0].sidecars)).toContain('GPU-corrected'); + } + const revision = await getGpuMetricsPointRevision(client(db), 10); + const writes = blob.counts.writes; + const unchanged = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: artifact(), + benchmarkResultIds: [10, 11], + }); + expect(unchanged.seriesSkipped).toBe(1); + expect(await getGpuMetricsPointRevision(client(db), 10)).toBe(revision); + await request(10); + expect(blob.counts.writes).toBe(writes); + const samples = await db.query('select count(*)::int as n from gpu_metric_samples'); + expect(samples.rows).toEqual([{ n: 24 }]); + }); +}); + +const power = (body: { series: GpuMetricSeries[] }) => + body.series[0]!.stats.find((s) => s.gpuIndex === 0 && s.metric === 'power_w')!; + +it('bypasses warmed shared-point caches on algorithm upgrade, then persists through unchanged re-ingest', async () => { + const input = { workflowRunId: 1, artifact: artifact(), benchmarkResultIds: [10, 11] }; + const first = await ingestGpuMetricsArtifact(sql, input); + await sql`update gpu_metric_gpu_stats set mean_value = -1`; + + for (const id of [10, 11]) { + const cached = await request(id); + expect(power(cached.body).mean).toBe(-1); + } + const oldRevision = await getGpuMetricsPointRevision(client(db), 10); + const originalSamples = + await sql`select * from gpu_metric_samples order by series_id, sampled_at, gpu_index`; + // Model a code deploy: no DB row, sample, sidecar or link has changed. + algorithm.version = 2; + expect(await getGpuMetricsPointRevision(client(db), 10)).not.toBe(oldRevision); + for (const id of [10, 11]) { + const refreshed = await request(id); + expect(power(refreshed.body).mean).toBeCloseTo(351.403333, 3); + } + expect(await sql`select stats_version from gpu_metric_series`).toEqual([{ stats_version: 1 }]); + expect(await sql`select distinct mean_value from gpu_metric_gpu_stats`).toEqual([ + { mean_value: -1 }, + ]); + const upgradedRevision = await getGpuMetricsPointRevision(client(db), 10); + const repaired = await ingestGpuMetricsArtifact(sql, input); + expect(repaired).toMatchObject({ + seriesIds: first.seriesIds, + samplesInserted: 0, + seriesSkipped: 0, + }); + expect(await getGpuMetricsPointRevision(client(db), 10)).not.toBe(upgradedRevision); + for (const id of [10, 11]) { + const refreshed = await request(id); + expect(power(refreshed.body).mean).toBeCloseTo(351.403333, 3); + } + expect(await sql`select stats_version from gpu_metric_series`).toEqual([{ stats_version: 2 }]); + expect( + await sql`select * from gpu_metric_samples order by series_id, sampled_at, gpu_index`, + ).toEqual(originalSamples); + const beforeHit = sampleReads; + await request(10); + expect(sampleReads).toBe(beforeHit); + expect(await ingestGpuMetricsArtifact(sql, input)).toMatchObject({ + samplesInserted: 0, + seriesSkipped: 1, + }); +}); diff --git a/packages/app/src/app/api/v1/gpu-metrics-point/route.ts b/packages/app/src/app/api/v1/gpu-metrics-point/route.ts new file mode 100644 index 000000000..6b995ecb2 --- /dev/null +++ b/packages/app/src/app/api/v1/gpu-metrics-point/route.ts @@ -0,0 +1,46 @@ +import { getDb } from '@semianalysisai/inferencex-db/connection'; +import { getGpuMetricsPointRevision } from '@semianalysisai/inferencex-db/queries/gpu-metrics-revision'; +import type { NextRequest } from 'next/server'; +import { + getGpuMetricsForPoint, + type GpuMetricsPointPayload, +} from '@semianalysisai/inferencex-db/queries/gpu-metrics'; + +import { cachedQuery } from '@/lib/api-cache'; + +import { idQueryRoute } from '../id-routes'; + +export const dynamic = 'force-dynamic'; + +export const CACHE_KEY_PREFIX = 'gpu-metrics-point-v2'; + +const getCachedGpuMetricsForPoint = cachedQuery( + (id: number, _revision: string): Promise => + getGpuMetricsForPoint(getDb(), id), + CACHE_KEY_PREFIX, + { blobOnly: true }, +); + +/** + * GET /api/v1/gpu-metrics-point?id=N + * + * PowerX telemetry recorded while one benchmark point ran: every linked + * gpu_metrics series with full-resolution per-GPU samples and the ingest-time + * per-GPU statistics digest. 404 when the point has no linked series (its + * run predates migration 016 and the artifacts have expired, or the job + * uploaded no gpu_metrics artifact). + */ +const handleGet = idQueryRoute({ + logLabel: 'gpu metrics point', + fetch: async (id) => { + const revision = await getGpuMetricsPointRevision(getDb(), id); + return revision === null ? null : getCachedGpuMetricsForPoint(id, revision); + }, +}); + +export async function GET(request: NextRequest): Promise { + const response = await handleGet(request); + // The live revision check must run even when a point used to be missing. + response.headers.set('Cache-Control', 'no-store'); + return response; +} diff --git a/packages/app/src/app/api/v1/views/extensions.test.ts b/packages/app/src/app/api/v1/views/extensions.test.ts index 11d6dac9a..99402904f 100644 --- a/packages/app/src/app/api/v1/views/extensions.test.ts +++ b/packages/app/src/app/api/v1/views/extensions.test.ts @@ -19,7 +19,7 @@ const mocks = vi.hoisted(() => ({ unofficial: vi.fn(), submissions: vi.fn(), })); -vi.mock('@/app/api/gpu-metrics/route', () => ({ GET: mocks.metrics })); +vi.mock('@/app/api/gpu-metrics/route', () => ({ readGpuMetricsForView: mocks.metrics })); vi.mock('@/app/api/video-runs/route', () => ({ GET: mocks.video })); vi.mock('@/app/api/v1/benchmarks/route', () => ({ GET: mocks.benchmarks })); vi.mock('@/app/api/unofficial-run/route', () => ({ GET: mocks.unofficial })); diff --git a/packages/app/src/app/api/v1/views/gpu-metrics/route.test.ts b/packages/app/src/app/api/v1/views/gpu-metrics/route.test.ts new file mode 100644 index 000000000..b4f3f167b --- /dev/null +++ b/packages/app/src/app/api/v1/views/gpu-metrics/route.test.ts @@ -0,0 +1,101 @@ +import type { GpuMetricStatRow } from '@semianalysisai/inferencex-db/queries/gpu-metrics'; +import type { GpuMetricsArtifact, GpuPowerApiResponse } from '@/components/gpu-power/types'; +import { NextRequest } from 'next/server'; +import { beforeEach, describe, expect, it, vi } from 'vitest'; + +const { metrics } = vi.hoisted(() => ({ metrics: vi.fn() })); +vi.mock('@/app/api/gpu-metrics/route', () => ({ + GET: metrics, + readGpuMetricsForView: metrics, +})); +// These source handlers are imported by the shared source module, but this view never calls them. +vi.mock('@/app/api/v1/benchmarks/route', () => ({ GET: vi.fn() })); +vi.mock('@/app/api/unofficial-run/route', () => ({ GET: vi.fn() })); + +import { GET } from './route'; + +const rows = [ + { timestamp: '2026-09-08T00:00:00Z', index: 0, power: 100, temperature: 40 }, + { timestamp: '2026-09-08T00:00:01Z', index: 0, power: 300, temperature: 42 }, + { timestamp: '2026-09-08T00:00:01Z', index: 1, power: 50, temperature: 38 }, +]; +const digest: GpuMetricStatRow[] = [ + { + gpuIndex: 0, + metric: 'power_w', + count: 10, + min: 0, + max: 800, + mean: 350, + median: 300, + p95: 750, + p99: 790, + stddev: 200, + }, + { + gpuIndex: 1, + metric: 'power_w', + count: 12, + min: 100, + max: 900, + mean: 700, + median: 750, + p95: 880, + p99: 898, + stddev: 80, + }, +]; +function artifact(stats: GpuMetricStatRow[] = digest): GpuMetricsArtifact { + return { + name: 'gpu_metrics_retained/host-a/gpu_metrics.csv', + data: rows, + series: { + id: 1, + artifactName: 'gpu_metrics_retained', + configKey: 'retained', + fileName: 'host-a/gpu_metrics.csv', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 22, + gpuCount: 2, + startedAt: '2026-09-08T00:00:00Z', + endedAt: '2026-09-08T00:00:10Z', + sidecars: {}, + benchmarkResultIds: [10, 11], + stats, + }, + }; +} +const runInfo = { + id: 34175132645, + name: 'Run Sweep', + branch: 'main', + sha: 'retained-source', + createdAt: '2026-09-08T00:00:00Z', + url: 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/34175132645', + conclusion: 'success', + status: 'completed', +}; +function source(artifacts: GpuMetricsArtifact[]) { + const payload: GpuPowerApiResponse = { runInfo, artifacts }; + metrics.mockImplementation(() => Response.json(payload)); +} +const request = (query = '') => + new NextRequest(`http://localhost/api/v1/views/gpu-metrics?runId=${runInfo.id}${query}`); + +beforeEach(() => { + vi.clearAllMocks(); + source([artifact()]); +}); + +describe('GET /api/v1/views/gpu-metrics full-record statistics', () => { + it('returns the stored digest even when raw rows would produce different statistics', async () => { + const response = await GET(request()); + expect(response.status).toBe(200); + expect(response.headers.get('cache-control')).toBe('private, no-store'); + const body = await response.json(); + expect(body.stats).toEqual(digest.map(({ metric: _metric, ...stats }) => stats)); + expect(body.rows).toEqual(rows); + expect(metrics.mock.calls[0][0].nextUrl.searchParams.get('runId')).toBe(String(runInfo.id)); + }); +}); diff --git a/packages/app/src/app/api/v1/views/gpu-metrics/route.ts b/packages/app/src/app/api/v1/views/gpu-metrics/route.ts index 253c8bb1f..9b9ae967d 100644 --- a/packages/app/src/app/api/v1/views/gpu-metrics/route.ts +++ b/packages/app/src/app/api/v1/views/gpu-metrics/route.ts @@ -1,10 +1,13 @@ -import { GET as metrics } from '@/app/api/gpu-metrics/route'; +import { + readGpuMetricsForView as metrics, + type GpuMetricsRouteResponse, +} from '@/app/api/gpu-metrics/route'; import { buildCorrelationData, buildGroupedData } from '@/components/gpu-power/chart-data'; +import { storedGpuStatsForMetric } from '@/components/gpu-power/stored-gpu-stats'; import { ALL_METRIC_OPTIONS, computeGpuStats, getAvailableMetrics, - type GpuPowerApiResponse, } from '@/components/gpu-power/types'; import { runViewsRoute, ViewsApiParamError } from '@/lib/views-api/errors'; import { @@ -21,24 +24,24 @@ import { type NextRequest, NextResponse } from 'next/server'; export const dynamic = 'force-dynamic'; export const maxDuration = 300; -/** Live metrics intentionally bypass both CDN and Blob caching. */ +/** Stored and live metrics bypass response caching so repairs remain observable. */ export function GET(request: NextRequest) { return runViewsRoute('gpu-metrics', async () => { validateViewParams(request.nextUrl.searchParams, VIEW_QUERY_PARAMS['gpu-metrics']); const s = request.nextUrl.searchParams; if (!s.get('runId')) throw new ViewsApiParamError('runId', 'runId is required'); const runId = parseNumberParam(s.get('runId'), 'runId', 0, { min: 1, integer: true }); - const data = await readResponse( - await metrics(sourceRequest(request, '/api/gpu-metrics', { runId: String(runId) })), + const data = await readResponse( + await metrics( + sourceRequest(request, '/api/gpu-metrics', { runId: String(runId) }), + s.get('artifact'), + ), ); + const artifactNames = data.artifactNames ?? data.artifacts.map((a) => a.name); const artifact = s.get('artifact') ?? data.artifacts[0]?.name; const selected = data.artifacts.find((a) => a.name === artifact); if (artifact && !selected) - throw new ViewsApiParamError( - 'artifact', - 'Unknown artifact', - data.artifacts.map((a) => a.name), - ); + throw new ViewsApiParamError('artifact', 'Unknown artifact', artifactNames); const rows = selected?.data ?? []; const availableMetrics = getAvailableMetrics(rows); const keys = ALL_METRIC_OPTIONS.map((m) => m.key); @@ -70,9 +73,11 @@ export function GET(request: NextRequest) { ); const direction = parseEnumParam(s.get('direction'), 'direction', ['asc', 'desc'], 'asc'); // The UI statistics table uses all chips; chart visibility does not filter it. - const stats = computeGpuStats(rows, metric).sort( - (a, b) => (a[sort] - b[sort]) * (direction === 'asc' ? 1 : -1), - ); + const stats = ( + selected?.series + ? storedGpuStatsForMetric(selected.series.stats, metric) + : computeGpuStats(rows, metric) + ).sort((a, b) => (a[sort] - b[sort]) * (direction === 'asc' ? 1 : -1)); return NextResponse.json( { view: 'gpu-metrics', @@ -90,7 +95,7 @@ export function GET(request: NextRequest) { direction, }, runInfo: data.runInfo, - artifacts: data.artifacts.map((a) => a.name), + artifacts: artifactNames, availableMetrics, gpuIndices: indices, rows: rows.filter((r) => gpus.includes(r.index)), diff --git a/packages/app/src/components/gpu-power/GpuPowerChart.tsx b/packages/app/src/components/gpu-power/GpuPowerChart.tsx index 09635c51a..fa3f33b5d 100644 --- a/packages/app/src/components/gpu-power/GpuPowerChart.tsx +++ b/packages/app/src/components/gpu-power/GpuPowerChart.tsx @@ -2,13 +2,25 @@ import * as d3 from 'd3'; import React, { useMemo } from 'react'; -import { buildGroupedData, type ParsedPoint } from './chart-data'; +import { buildTelemetryData, type ParsedPoint } from './chart-data'; import { D3Chart } from '@/lib/d3-chart/D3Chart'; +import type { RenderContext } from '@/lib/d3-chart/D3Chart/types'; +import { CHART_TYPE, px } from '@/lib/d3-chart/typography'; import { useLocale } from '@/lib/use-locale'; +import { + DEFAULT_TELEMETRY_DISPLAY, + estimateSampleIntervalMs, + meanAcrossSeries, + nearestSample, + rollingTimeAverage, + type TelemetryDisplayState, + type TimedSample, +} from './telemetry-smoothing'; import { ALL_METRIC_OPTIONS, detectTdpFromArtifactName, + tdpForHardware, getGpuMetricLabel, getGpuMetricYAxisLabel, type GpuMetricKey, @@ -26,6 +38,10 @@ const STRINGS = { power: 'Power', temp: 'Temp', utilization: 'Chip Util', + rollingSuffix: (windowS: number) => `${windowS} s rolling avg`, + meanChips: 'Mean of visible chips', + meanCount: (n: number) => `${n} chips`, + rightAxis: 'right axis', }, zh: { empty: '暂无可显示的芯片指标数据。', @@ -36,23 +52,155 @@ const STRINGS = { power: '功耗', temp: '温度', utilization: '芯片利用率', + rollingSuffix: (windowS: number) => `${windowS} 秒滚动平均`, + meanChips: '可见芯片均值', + meanCount: (n: number) => `${n} 个芯片`, + rightAxis: '右轴', }, } as const; +/** A second time series drawn over the telemetry on its own right-hand axis. */ +export interface TelemetryOverlaySeries { + label: string; + unit: string; + color: string; + /** Absolute-time samples; the chart re-bases them onto its own t=0. */ + points: TimedSample[]; +} + interface GpuMetricsChartProps { data: GpuMetricRow[]; visibleGpus: Set; metricKey: GpuMetricKey; artifactName: string; + /** + * Hardware key of the benchmark point (`b200`, `h200`, …). Multinode bundle + * names are hash-truncated and carry no SKU token, so callers that know the + * point pass it explicitly; the artifact-name sniff stays the fallback. + */ + hardware?: string; legendElement?: React.ReactNode; caption?: React.ReactNode; /** Max interactive points before LTTB downsampling. Infinity to disable. */ maxPoints?: number; + /** Raw samples vs. time-window rolling average. Defaults to raw samples. */ + display?: TelemetryDisplayState; + /** Optional secondary series (e.g. decode throughput) on a right y-axis. */ + overlay?: TelemetryOverlaySeries | null; +} + +/** Mean across the visible chips, aligned by nearest sample within one poll interval. */ +function buildMeanSeries(groups: Map, t0Ms: number): ParsedPoint[] { + const arrays = [...groups.values()]; + if (arrays.length === 0) return []; + const longest = arrays.reduce((a, b) => (b.length > a.length ? b : a)); + const mean = meanAcrossSeries(arrays, estimateSampleIntervalMs(longest)); + return mean.map((p) => ({ + seconds: (p.ms - t0Ms) / 1000, + ms: p.ms, + value: p.value, + gpuIndex: MEAN_INDEX, + raw: null, + count: p.count, + })); +} + +/** Replace each point's value with its centered time-window mean. */ +function smoothSeries(points: ParsedPoint[], windowS: number): ParsedPoint[] { + const averaged = rollingTimeAverage(points, windowS * 1000); + return points.map((point, i) => ({ ...point, value: averaged[i]!.value, raw: null })); +} + +/** Per-chip palette shared with legends that toggle chips on and off. */ +export const GPU_COLORS = d3.schemeTableau10; +/** Pseudo chip index and line key for the mean across visible chips. */ +const MEAN_INDEX = -1; +const MEAN_KEY = 'mean'; +const MEAN_COLOR = 'var(--foreground)'; + +function lineKey(gpuIndex: number): string { + return gpuIndex === MEAN_INDEX ? MEAN_KEY : String(gpuIndex); } -const GPU_COLORS = d3.schemeTableau10; +function colorFor(gpuIndex: number): string { + return gpuIndex === MEAN_INDEX ? MEAN_COLOR : GPU_COLORS[gpuIndex % GPU_COLORS.length]!; +} const CHART_ID = 'gpu-metrics-line'; const MARGIN = { top: 24, right: 20, bottom: 60, left: 60 }; +/** Room for the overlay's right axis ticks and rotated title. */ +const MARGIN_WITH_OVERLAY = { ...MARGIN, right: 84 }; + +interface OverlayPoint { + x: number; + y: number; +} + +function overlayYScale(points: OverlayPoint[], height: number): d3.ScaleLinear { + const max = d3.max(points, (p) => p.y) ?? 0; + return d3 + .scaleLinear() + .domain([0, max > 0 ? max * 1.05 : 1]) + .range([height, 0]) + .nice(); +} + +function overlayLine( + xScale: d3.ScaleLinear, + yScale: d3.ScaleLinear, +): d3.Line { + return d3 + .line() + .x((p) => xScale(p.x)) + .y((p) => yScale(p.y)) + .curve(d3.curveMonotoneX); +} + +/** + * Draw the overlay path inside the clipped zoom group and its axis in the + * unclipped root group. Both are removed first so toggling the overlay off + * (or re-rendering) never leaves a stale axis behind. + */ +function renderOverlay( + group: d3.Selection, + ctx: RenderContext, + overlay: TelemetryOverlaySeries | null | undefined, + points: OverlayPoint[], +): void { + group.selectAll('.telemetry-overlay').remove(); + ctx.layout.g.selectAll('.telemetry-overlay-axis').remove(); + if (!overlay || points.length === 0) return; + const xScale = ctx.xScale as d3.ScaleLinear; + const yScale = overlayYScale(points, ctx.height); + group + .append('path') + .attr('class', 'telemetry-overlay') + .attr('fill', 'none') + .attr('stroke', overlay.color) + .attr('stroke-width', 1.75) + .attr('opacity', 0.9) + .attr('pointer-events', 'none') + .attr('d', overlayLine(xScale, yScale)(points)); + + const axis = ctx.layout.g + .append('g') + .attr('class', 'telemetry-overlay-axis') + .attr('transform', `translate(${ctx.width},0)`) + .call(d3.axisRight(yScale).ticks(6).tickSize(4).tickFormat(d3.format('~s'))); + axis.select('.domain').attr('stroke', overlay.color); + axis.selectAll('.tick line').attr('stroke', overlay.color); + axis + .selectAll('.tick text') + .attr('fill', overlay.color) + .attr('font-size', px(CHART_TYPE.axisLabel)); + axis + .append('text') + .attr('class', 'telemetry-overlay-axis-label') + .attr('transform', `translate(${ctx.layout.margin.right - 14},${ctx.height / 2}) rotate(90)`) + .attr('text-anchor', 'middle') + .attr('fill', overlay.color) + .attr('font-size', px(CHART_TYPE.axisLabel)) + .text(`${overlay.label} (${overlay.unit})`); +} const GpuMetricsChart = React.memo( ({ @@ -60,19 +208,42 @@ const GpuMetricsChart = React.memo( visibleGpus, metricKey, artifactName, + hardware, legendElement, caption, maxPoints, + display = DEFAULT_TELEMETRY_DISPLAY, + overlay, }: GpuMetricsChartProps) => { const locale = useLocale(); const t = STRINGS[locale]; const metricConfig = ALL_METRIC_OPTIONS.find((m) => m.key === metricKey)!; + const rolling = display.mode === 'rolling'; + const showChips = display.series !== 'mean'; + const showMean = display.series !== 'chips'; - const groupedData = useMemo( - () => buildGroupedData(data, visibleGpus, metricKey), + const { t0Ms, groups: rawGroups } = useMemo( + () => buildTelemetryData(data, visibleGpus, metricKey), [data, visibleGpus, metricKey], ); + // Displayed series keyed by chip index (MEAN_INDEX for the mean line): + // per-chip and/or mean, then optionally smoothed. + const groupedData = useMemo(() => { + const series = new Map(); + if (showChips) for (const [gpuIndex, points] of rawGroups) series.set(gpuIndex, points); + if (showMean) { + const mean = buildMeanSeries(rawGroups, t0Ms); + if (mean.length > 0) series.set(MEAN_INDEX, mean); + } + if (!rolling) return series; + const smoothed = new Map(); + for (const [gpuIndex, points] of series) { + smoothed.set(gpuIndex, smoothSeries(points, display.windowS)); + } + return smoothed; + }, [rawGroups, t0Ms, showChips, showMean, rolling, display.windowS]); + const allPoints = useMemo(() => { const pts: ParsedPoint[] = []; for (const points of groupedData.values()) pts.push(...points); @@ -83,11 +254,64 @@ const GpuMetricsChart = React.memo( const lineData = useMemo(() => { const result: Record = {}; for (const [gpuIndex, points] of groupedData) { - result[String(gpuIndex)] = points.map((p) => ({ x: p.seconds, y: p.value })); + result[lineKey(gpuIndex)] = points.map((p) => ({ x: p.seconds, y: p.value })); } return result; }, [groupedData]); + // Overlay samples, smoothed with the same window as the chip lines when + // averaging so both series answer the same "how much over N seconds" + // question, then re-based onto the telemetry's t=0. + const overlaySamples = useMemo(() => { + if (!overlay) return []; + return rolling ? rollingTimeAverage(overlay.points, display.windowS * 1000) : overlay.points; + }, [overlay, rolling, display.windowS]); + const overlayPoints = useMemo( + () => overlaySamples.map((p) => ({ x: (p.ms - t0Ms) / 1000, y: p.value })), + [overlaySamples, t0Ms], + ); + const hasOverlay = Boolean(overlay) && overlayPoints.length > 0; + + const hasMeanLine = groupedData.has(MEAN_INDEX); + const keyRow = + hasMeanLine || hasOverlay ? ( +
+ {hasMeanLine && ( + + + {t.meanChips} + + )} + {hasOverlay && overlay && ( + + + {overlay.label} ({overlay.unit} · {t.rightAxis}) + + )} +
+ ) : null; + const resolvedCaption = + caption || keyRow ? ( + <> + {caption} + {keyRow} + + ) : undefined; + // Scale domains const xDomain = useMemo(() => { if (allPoints.length === 0) return [0, 100] as [number, number]; @@ -95,14 +319,23 @@ const GpuMetricsChart = React.memo( return ext; }, [allPoints]); - const tdpInfo = metricKey === 'power' ? detectTdpFromArtifactName(artifactName) : null; + const tdpInfo = + metricKey === 'power' + ? (tdpForHardware(hardware) ?? detectTdpFromArtifactName(artifactName)) + : null; const yDomain = useMemo(() => { if (allPoints.length === 0) return [0, 100] as [number, number]; const ext = d3.extent(allPoints, (d) => d.value) as [number, number]; const range = ext[1] - ext[0]; - const yMin = Math.max(0, ext[0] - range * 0.05); - let yMax = ext[1] + range * 0.05; + // Constant telemetry and floating-point noise still need a readable axis. + const padding = Math.max( + range * 0.05, + Math.max(Math.abs(ext[0]), Math.abs(ext[1])) * 0.01, + 1, + ); + const yMin = Math.max(0, ext[0] - padding); + let yMax = ext[1] + padding; if (tdpInfo && tdpInfo.tdp > yMax) yMax = tdpInfo.tdp * 1.05; return [yMin, yMax] as [number, number]; }, [allPoints, tdpInfo]); @@ -120,46 +353,52 @@ const GpuMetricsChart = React.memo( chartId={CHART_ID} data={allPoints} height={600} - margin={MARGIN} + margin={hasOverlay ? MARGIN_WITH_OVERLAY : MARGIN} watermark="logo" testId="gpu-metrics-chart-svg" grabCursor={true} instructions={t.instructions} xScale={{ type: 'linear', domain: xDomain, nice: true }} yScale={{ type: 'linear', domain: yDomain, nice: true }} - xAxis={{ label: t.seconds, tickCount: 10 }} + xAxis={{ + label: t.seconds, + tickValues: (scale) => { + const secondsScale = scale as d3.ScaleLinear; + const [left, right] = secondsScale.range(); + return secondsScale.ticks(Math.max(2, Math.min(10, Math.floor((right - left) / 60)))); + }, + }} yAxis={{ label: getGpuMetricYAxisLabel(metricConfig, locale), tickCount: 8 }} layers={[ // TDP reference line (power metric only) { type: 'custom', key: 'tdp-line', - render: tdpInfo - ? (group, ctx) => { - const yScale = ctx.yScale as d3.ScaleLinear; - const tdpY = yScale(tdpInfo.tdp); - group.selectAll('.tdp-line').remove(); - const tdpGroup = group.append('g').attr('class', 'tdp-line'); - tdpGroup - .append('line') - .attr('x1', 0) - .attr('x2', ctx.width) - .attr('y1', tdpY) - .attr('y2', tdpY) - .attr('stroke', '#ef4444') - .attr('stroke-width', 1.5) - .attr('stroke-dasharray', '6,4'); - tdpGroup - .append('text') - .attr('x', ctx.width - 4) - .attr('y', tdpY - 6) - .attr('text-anchor', 'end') - .attr('fill', '#ef4444') - .attr('font-size', '11px') - .attr('font-weight', '600') - .text(`${tdpInfo.sku} TDP: ${tdpInfo.tdp}W`); - } - : null, + render: (group, ctx) => { + group.selectAll('.tdp-line').remove(); + if (!tdpInfo) return; + const yScale = ctx.yScale as d3.ScaleLinear; + const tdpY = yScale(tdpInfo.tdp); + const tdpGroup = group.append('g').attr('class', 'tdp-line'); + tdpGroup + .append('line') + .attr('x1', 0) + .attr('x2', ctx.width) + .attr('y1', tdpY) + .attr('y2', tdpY) + .attr('stroke', '#ef4444') + .attr('stroke-width', 1.5) + .attr('stroke-dasharray', '6,4'); + tdpGroup + .append('text') + .attr('x', ctx.width - 4) + .attr('y', tdpY - 6) + .attr('text-anchor', 'end') + .attr('fill', '#ef4444') + .attr('font-size', '11px') + .attr('font-weight', '600') + .text(`${tdpInfo.sku} TDP: ${tdpInfo.tdp}W`); + }, }, // GPU lines { @@ -167,8 +406,8 @@ const GpuMetricsChart = React.memo( key: 'gpu-lines', lines: lineData, config: { - getColor: (key) => GPU_COLORS[parseInt(key, 10) % GPU_COLORS.length], - strokeWidth: 1.5, + getColor: (key) => (key === MEAN_KEY ? MEAN_COLOR : colorFor(parseInt(key, 10))), + getStrokeWidth: (key) => (key === MEAN_KEY ? 2.5 : rolling ? 1.75 : 1.5), curve: d3.curveMonotoneX, }, }, @@ -182,11 +421,31 @@ const GpuMetricsChart = React.memo( getCy: () => 0, getX: (d) => d.seconds, getY: (d) => d.value, - getColor: (d) => GPU_COLORS[d.gpuIndex % GPU_COLORS.length], - getRadius: () => 2, + // Averaged mode draws lines only; the circles stay as invisible + // hover targets so the tooltip and crosshair keep working. + getColor: (d) => (rolling ? 'transparent' : colorFor(d.gpuIndex)), + getRadius: () => (rolling ? 3 : 2), maxPoints, }, }, + // Secondary series on a right-hand axis (e.g. decode throughput) + { + type: 'custom', + key: 'telemetry-overlay', + render: (group, ctx) => { + renderOverlay(group, ctx, hasOverlay ? overlay : null, overlayPoints); + }, + onZoom: (group, ctx) => { + if (!hasOverlay) return; + const xScale = ctx.newXScale as d3.ScaleLinear; + group + .select('.telemetry-overlay') + .attr( + 'd', + overlayLine(xScale, overlayYScale(overlayPoints, ctx.height))(overlayPoints), + ); + }, + }, ]} zoom={{ enabled: true, @@ -197,29 +456,60 @@ const GpuMetricsChart = React.memo( tooltip={{ rulerType: 'crosshair', content: (d: ParsedPoint, isPinned: boolean) => { - const color = GPU_COLORS[d.gpuIndex % GPU_COLORS.length]; + const color = colorFor(d.gpuIndex); + const sep = locale === 'zh' ? ':' : ':'; + const title = + d.gpuIndex === MEAN_INDEX + ? `${t.meanChips}${d.count ? ` · ${t.meanCount(d.count)}` : ''}` + : `${t.chip} ${d.gpuIndex}`; + const overlayAt = hasOverlay ? nearestSample(overlaySamples, d.ms) : null; + const overlayRow = + overlay && overlayAt + ? `
${overlay.label}${sep} ${ + overlayAt.value >= 100 ? overlayAt.value.toFixed(0) : overlayAt.value.toFixed(1) + } ${overlay.unit}
` + : ''; return `
${isPinned ? `
${t.dismiss}
` : ''} -
${t.chip} ${d.gpuIndex}
+
${title}
${d.seconds.toFixed(1)}${locale === 'zh' ? ' 秒' : 's'}
-
${getGpuMetricLabel(metricConfig, locale)}${locale === 'zh' ? ':' : ':'} ${d.value.toFixed(1)} ${metricConfig.unit}
-
${t.power}${locale === 'zh' ? ':' : ':'} ${d.raw.power.toFixed(1)} W
-
${t.temp}${locale === 'zh' ? ':' : ':'} ${d.raw.temperature}\u00B0C
-
${t.utilization}${locale === 'zh' ? ':' : ':'} ${d.raw.gpuUtil}%
+
${getGpuMetricLabel(metricConfig, locale)}${sep} ${d.value.toFixed(1)} ${metricConfig.unit}
+ ${rolling ? `
${t.rollingSuffix(display.windowS)}
` : ''} + ${ + d.raw + ? `
${t.power}${sep} ${d.raw.power.toFixed(1)} W
${ + d.raw.temperature === undefined + ? '' + : `
${t.temp}${sep} ${d.raw.temperature}\u00B0C
` + }${ + d.raw.gpuUtil === undefined + ? '' + : `
${t.utilization}${sep} ${d.raw.gpuUtil}%
` + }` + : '' + } + ${overlayRow}
`; }, getRulerX: (d, xScale) => (xScale as d3.ScaleLinear)(d.seconds), getRulerY: (d, yScale) => yScale(d.value), - onHoverStart: (sel) => { - sel.attr('r', 5).attr('stroke', 'white').attr('stroke-width', 1); + onHoverStart: (sel, d) => { + sel + .attr('r', 5) + .attr('fill', colorFor(d.gpuIndex)) + .attr('stroke', 'white') + .attr('stroke-width', 1); }, - onHoverEnd: (sel) => { - sel.attr('r', 2).attr('stroke', 'none'); + onHoverEnd: (sel, d) => { + sel + .attr('r', rolling ? 3 : 2) + .attr('fill', rolling ? 'transparent' : colorFor(d.gpuIndex)) + .attr('stroke', 'none'); }, attachToLayer: 2, }} legendElement={legendElement} - caption={caption} + caption={resolvedCaption} /> ); }, diff --git a/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx b/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx index f0949589c..d28dbf25d 100644 --- a/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx +++ b/packages/app/src/components/gpu-power/GpuPowerDisplay.tsx @@ -33,6 +33,8 @@ import { useClientSearchParams } from '@/hooks/useClientSearch'; import GpuCorrelationChart from './GpuCorrelationChart'; import GpuMetricsChart from './GpuPowerChart'; import GpuStatsTable from './GpuStatsTable'; +import { DEFAULT_TELEMETRY_DISPLAY, type TelemetryDisplayState } from './telemetry-smoothing'; +import { TelemetryDisplayControls } from './TelemetryDisplayControls'; import { type GpuMetricKey, type GpuPowerApiResponse, @@ -81,6 +83,7 @@ const STRINGS = { chip: 'Chip', chartToolbar: 'Chart controls', correlationAxes: 'Correlation axes', + displayControls: 'Line options', }, zh: { heading: 'PowerX', @@ -119,6 +122,7 @@ const STRINGS = { chip: '芯片', chartToolbar: '图表控制', correlationAxes: '相关性坐标轴', + displayControls: '曲线选项', }, } as const; @@ -175,6 +179,7 @@ export default function GpuMetricsDisplay() { gcTime: 0, retry: false, refetchOnMount: 'always', + refetchOnWindowFocus: true, }); const artifacts = query.data?.artifacts ?? []; const runInfo = query.data?.runInfo ?? null; @@ -200,10 +205,8 @@ export default function GpuMetricsDisplay() { artifacts.some((artifact) => artifact.name === selectedArtifactCandidate) ? selectedArtifactCandidate : (artifacts[0]?.name ?? ''); - const currentData = useMemo( - () => artifacts.find((artifact) => artifact.name === selectedArtifact)?.data ?? [], - [artifacts, selectedArtifact], - ); + const currentArtifact = artifacts.find((artifact) => artifact.name === selectedArtifact); + const currentData = useMemo(() => currentArtifact?.data ?? [], [currentArtifact]); const availableMetrics = useMemo(() => getAvailableMetrics(currentData), [currentData]); const urlMetric = searchParams.get('gm_metric'); const selectedMetricCandidate = selectionApplies ? selection.metric : urlMetric; @@ -237,6 +240,16 @@ export default function GpuMetricsDisplay() { const [chartView, setChartView] = useState('chart'); const [corrXMetric, setCorrXMetric] = useState('power'); const [corrYMetric, setCorrYMetric] = useState('temperature'); + // A power-only series (multinode DCGM bundle) has no temperature axis to + // default to; use the first other collected metric instead of an empty plot. + const effectiveCorrYMetric = useMemo( + () => + availableMetrics.some((m) => m.key === corrYMetric) + ? corrYMetric + : (availableMetrics.find((m) => m.key !== corrXMetric)?.key ?? corrXMetric), + [availableMetrics, corrXMetric, corrYMetric], + ); + const [display, setDisplay] = useState(DEFAULT_TELEMETRY_DISPLAY); const viewOptions = useMemo[]>( () => [ { @@ -359,6 +372,52 @@ export default function GpuMetricsDisplay() { else setCorrYMetric(value as GpuMetricKey); }, []); + const legendElement = ( + ({ + name: `${t.chip} ${gpuIndex}`, + hw: String(gpuIndex), + label: `${t.chip} ${gpuIndex}`, + color: GPU_COLORS[gpuIndex % GPU_COLORS.length], + isActive: visibleGpus.has(gpuIndex), + onClick: () => toggleGpu(gpuIndex), + }))} + isLegendExpanded={isLegendExpanded} + onExpandedChange={(expanded) => { + setIsLegendExpanded(expanded); + track('gpu_metrics_legend_expanded', { expanded }); + }} + actions={ + allGpusSelected + ? [] + : [ + { + id: + chartView === 'correlation' + ? 'gpu-metrics-reset-filter-2' + : 'gpu-metrics-reset-filter', + label: t.resetFilter, + onClick: selectAllGpus, + }, + ] + } + switches={[ + { + id: + chartView === 'correlation' ? 'gpu-metrics-downsample-corr' : 'gpu-metrics-downsample', + label: t.downsample, + checked: downsample, + onCheckedChange: (c) => { + setDownsample(c); + track('gpu_metrics_downsample_toggled', { enabled: c }); + }, + }, + ]} + /> + ); + return (
@@ -608,7 +667,7 @@ export default function GpuMetricsDisplay() {
{ + const windowS = Number(raw) as SmoothingWindowS; + track(`${analyticsPrefix}_smoothing_window_changed`, { windowS }); + onChange({ ...value, windowS }); + }} + > + + + + + {SMOOTHING_WINDOWS_S.map((s) => ( + + {t.windowOption(s)} + + ))} + + +
+ )} +
+ + { + track(`${analyticsPrefix}_series_mode_changed`, { series }); + onChange({ ...value, series }); + }} + /> +
+ + + ); +} diff --git a/packages/app/src/components/gpu-power/chart-data.test.ts b/packages/app/src/components/gpu-power/chart-data.test.ts new file mode 100644 index 000000000..f7912f47c --- /dev/null +++ b/packages/app/src/components/gpu-power/chart-data.test.ts @@ -0,0 +1,33 @@ +import { describe, expect, it } from 'vitest'; +import { buildCorrelationData, buildGroupedData, buildTelemetryData } from './chart-data'; +import type { GpuMetricRow } from './types'; + +const rows: GpuMetricRow[] = [ + { timestamp: '2026-09-21T00:00:00Z', index: 0, power: 100 }, + { timestamp: '2026-09-21T00:00:02Z', index: 1, power: 200, temperature: 40 }, + { timestamp: '2026-09-21T00:00:03Z', index: 1, power: 250 }, + { timestamp: '2026-09-21T00:00:04Z', index: 1, power: 0, temperature: 0 }, +]; + +describe('shared telemetry chart preparation', () => { + it('keeps the whole-series time origin when a chip is hidden and skips missing readings', () => { + const { t0Ms, groups } = buildTelemetryData(rows, new Set([1]), 'temperature'); + expect(t0Ms).toBe(Date.parse(rows[0].timestamp)); + expect(groups.get(1)?.map(({ seconds, ms, value }) => ({ seconds, ms, value }))).toEqual([ + { seconds: 2, ms: t0Ms + 2000, value: 40 }, + { seconds: 4, ms: t0Ms + 4000, value: 0 }, + ]); + expect(groups.has(0)).toBe(false); + expect(buildGroupedData(rows, new Set([1]), 'temperature').get(1)).toEqual([ + { seconds: 2, value: 40, gpuIndex: 1, raw: rows[1] }, + { seconds: 4, value: 0, gpuIndex: 1, raw: rows[3] }, + ]); + }); + + it('omits unsampled correlation metrics while retaining measured zeros', () => { + expect(buildCorrelationData(rows, new Set([0, 1]), 'power', 'temperature')).toEqual([ + { x: 200, y: 40, gpuIndex: 1, raw: rows[1] }, + { x: 0, y: 0, gpuIndex: 1, raw: rows[3] }, + ]); + }); +}); diff --git a/packages/app/src/components/gpu-power/chart-data.ts b/packages/app/src/components/gpu-power/chart-data.ts index 7c0aef074..51db731f9 100644 --- a/packages/app/src/components/gpu-power/chart-data.ts +++ b/packages/app/src/components/gpu-power/chart-data.ts @@ -1,10 +1,16 @@ import type { GpuMetricKey, GpuMetricRow } from './types'; export interface ParsedPoint { seconds: number; + /** Absolute sample time in ms; smoothing and alignment work in this space. */ + ms: number; value: number; gpuIndex: number; - raw: GpuMetricRow; + /** The raw sample behind this point; null once the value has been averaged. */ + raw: GpuMetricRow | null; + /** For the mean line: how many chips contributed at this timestamp. */ + count?: number; } + function parseTimestamp(raw: string): Date | null { const isoDate = new Date(raw); if (!isNaN(isoDate.getTime())) return isoDate; @@ -15,28 +21,33 @@ function parseTimestamp(raw: string): Date | null { return null; } -export function buildGroupedData( +export function buildTelemetryData( data: GpuMetricRow[], visibleGpus: Set, metricKey: GpuMetricKey, -): Map { +): { t0Ms: number; groups: Map } { + // t=0 is the first sample of the whole series, not of the visible chips, so + // hiding a chip never shifts the time axis under the remaining lines. let minTime = Infinity; const parsed: { row: GpuMetricRow; ms: number }[] = []; for (const row of data) { - if (!visibleGpus.has(row.index)) continue; const time = parseTimestamp(row.timestamp); if (!time) continue; const ms = time.getTime(); - parsed.push({ row, ms }); if (ms < minTime) minTime = ms; + if (visibleGpus.has(row.index)) parsed.push({ row, ms }); } const groups = new Map(); for (const { row, ms } of parsed) { + const value = row[metricKey]; + // A metric the collector never sampled has no point, not a zero. + if (value === undefined) continue; if (!groups.has(row.index)) groups.set(row.index, []); groups.get(row.index)!.push({ seconds: (ms - minTime) / 1000, - value: row[metricKey] ?? 0, + ms, + value, gpuIndex: row.index, raw: row, }); @@ -44,7 +55,19 @@ export function buildGroupedData( for (const points of groups.values()) { points.sort((a, b) => a.seconds - b.seconds); } - return groups; + return { t0Ms: minTime, groups }; +} + +/** Keep the read-only view's serialized point shape while sharing chart preparation. */ +export function buildGroupedData( + data: GpuMetricRow[], + visibleGpus: Set, + metricKey: GpuMetricKey, +) { + const { groups } = buildTelemetryData(data, visibleGpus, metricKey); + return new Map( + [...groups].map(([index, points]) => [index, points.map(({ ms: _ms, ...point }) => point)]), + ); } export function buildCorrelationData( @@ -55,5 +78,9 @@ export function buildCorrelationData( ) { return data .filter((r) => visibleGpus.has(r.index)) - .map((r) => ({ x: r[xMetric] ?? 0, y: r[yMetric] ?? 0, gpuIndex: r.index, raw: r })); + .flatMap((r) => { + const x = r[xMetric]; + const y = r[yMetric]; + return x === undefined || y === undefined ? [] : [{ x, y, gpuIndex: r.index, raw: r }]; + }); } diff --git a/packages/app/src/components/gpu-power/power-audit-bundle.test.ts b/packages/app/src/components/gpu-power/power-audit-bundle.test.ts new file mode 100644 index 000000000..eb41678c5 --- /dev/null +++ b/packages/app/src/components/gpu-power/power-audit-bundle.test.ts @@ -0,0 +1,164 @@ +import { describe, expect, it } from 'vitest'; + +import { + BUNDLE_MANIFEST_ENTRY, + BUNDLE_SAMPLES_ENTRY, + BUNDLE_WINDOW_PAD_SECONDS, + cutPowerAuditBundle, + parsePowerCsvData, +} from './power-audit-bundle'; + +const ARTIFACT = 'power_audit_qwen3.5_8k1k_fp8_dynamo-sglang_x'; +const RESULT = 'qwen3.5_8k1k_fp8_dynamo-sglang_x_sa-bench_isl_8192_osl_1024'; + +/** `/` ids of the 2 × 2 synthetic cluster. */ +const A0 = 'cn01/GPU-a0'; +const A1 = 'cn01/GPU-a1'; +const B0 = 'cn02/GPU-b0'; +const B1 = 'cn02/GPU-b1'; + +const HEADER = 'schema_version,timestamp_unix,scrape_seq,hostname,gpu_index,gpu_uuid,power_w'; + +function sample(time: number, device: string, power: number): string { + const [hostname, uuid] = device.split('/'); + const gpuIndex = uuid.endsWith('0') ? 0 : 1; + return `1,${time},7,${hostname},${gpuIndex},${uuid},${power}`; +} + +/** Every device sampled once per whole second in `[from, to]`, in scrambled device order. */ +function tick(time: number, watts: Record): string[] { + return [B1, A1, B0, A0].map((device) => sample(time, device, watts[device])); +} + +const WATTS = { [A0]: 100, [A1]: 110, [B0]: 200, [B1]: 50 }; + +/** Window 1 at 1000–1010 s, window 2 at 2000–2010 s; nothing near window 3 at 5000 s. */ +const SAMPLES = [ + HEADER, + // Just outside the 60 s pad of window 1. + ...tick(1000 - BUNDLE_WINDOW_PAD_SECONDS - 1, WATTS), + // Pad boundaries and the window itself. + ...tick(1000 - BUNDLE_WINDOW_PAD_SECONDS, WATTS), + ...tick(1000, WATTS), + ...tick(1005.4, { [A0]: 300, [A1]: 310, [B0]: 400, [B1]: 60 }), + ...tick(1010, WATTS), + ...tick(1010 + BUNDLE_WINDOW_PAD_SECONDS, WATTS), + ...tick(1010 + BUNDLE_WINDOW_PAD_SECONDS + 1, WATTS), + // Window 2, sampled by three devices only. + sample(2000, A1, 120), + sample(2000, A0, 130), + sample(2000, B0, 220), + sample(2001, A1, 121), + sample(2001, A0, 131), + sample(2001, B0, 221), + // Malformed rows: short line, non-numeric power, blank power (`Number('')` + // is 0), blank gpu_index, blank line. + '1,2002,7,cn01', + `1,2002,7,cn01,0,GPU-a0,N/A`, + `1,2002,7,cn01,0,GPU-a0,`, + `1,2002,7,cn01, ,GPU-a0,55`, + '', +].join('\n'); + +const MANIFEST = JSON.stringify({ + schema_version: 1, + producer: 'srt-slurm.dcgm-power', + expected_devices: [ + { hostname: 'cn01', gpu_index: 0, assignments: [{ worker_role: 'prefill' }] }, + { hostname: 'cn01', gpu_index: 1, assignments: [{ worker_role: 'prefill' }] }, + { hostname: 'cn02', gpu_index: 0, assignments: [{ worker_role: 'decode' }] }, + { hostname: 'cn02', gpu_index: 1, assignments: [] }, + ], +}); + +function validationName(conc: number): string { + return `power_validation_${RESULT}_conc${conc}_gpus_4_ctx_2_gen_2.json`; +} + +/** Window 1: roles from `per_gpu_role`, where cn02/GPU-b0 is a prefill worker and GPU-b1 is unassigned. */ +const VALIDATION_1 = JSON.stringify({ + schema_version: 1, + benchmark_window: { start_time_unix: 900, end_time_unix: 1100 }, + selected_window: { concurrency: 1, start_time_unix: 1000, end_time_unix: 1010 }, + per_gpu_role: { [A0]: 'decode', [A1]: 'prefill', [B0]: 'prefill' }, +}); + +/** Window 2: no `per_gpu_role` and no `selected_window`, so the manifest and `benchmark_window` apply. */ +const VALIDATION_2 = JSON.stringify({ + schema_version: 1, + benchmark_window: { start_time_unix: 2000, end_time_unix: 2010 }, +}); + +/** Window 3: valid, but no sample lands within the pad. */ +const VALIDATION_3 = JSON.stringify({ + selected_window: { concurrency: 3, start_time_unix: 5000, end_time_unix: 5010 }, +}); + +function bundle(overrides: Record = {}): Map { + const files = new Map([ + [validationName(3), VALIDATION_3], + [validationName(1), VALIDATION_1], + [BUNDLE_SAMPLES_ENTRY, SAMPLES], + [BUNDLE_MANIFEST_ENTRY, MANIFEST], + [validationName(2), VALIDATION_2], + ]); + for (const [name, text] of Object.entries(overrides)) files.set(name, text); + const present = new Map(); + for (const [name, text] of files) if (text !== undefined) present.set(name, text); + return present; +} + +describe('cutPowerAuditBundle', () => { + it('orders rows prefill, decode, unassigned, then by hostname and gpu_index', () => { + const [first] = cutPowerAuditBundle(ARTIFACT, bundle()); + expect(first.devices).toEqual([ + { id: A1, role: 'prefill' }, + { id: B0, role: 'prefill' }, + { id: A0, role: 'decode' }, + { id: B1 }, + ]); + expect(first.gpus).toEqual([0, 1, 2, 3]); + expect(first.power).toHaveLength(4); + // Row order is the device order: A1 reads 110 W, then 310 W in the window. + expect(first.power[0][1]).toBe(110); + expect(first.power[0][2]).toBe(310); + expect(first.power[3][2]).toBe(60); + }); + + it('clips samples to the window plus the pad on both sides', () => { + const [first] = cutPowerAuditBundle(ARTIFACT, bundle()); + expect(first.startMs).toBe((1000 - BUNDLE_WINDOW_PAD_SECONDS) * 1000); + expect(first.t).toEqual([0, 60, 65, 70, 130]); + // 1005.4 s lands in the 1005 s bucket; the malformed 2002 s rows never appear. + expect(first.power[2]).toEqual([100, 100, 300, 100, 100]); + }); +}); + +describe('parsePowerCsvData', () => { + const NVIDIA_CSV = [ + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]', + '2026/09/11 04:19:41.123, 0, 350.5 W, 60, 1980 MHz, 2619 MHz, 90 %, 50 %', + ].join('\n'); + + it('normalizes UTC and missing collector context to ISO UTC like the stored path', () => { + // A zone-less stamp left as-is would be parsed as browser-local by `new Date`. + for (const context of [null, { timestamp_timezone: 'UTC' }, { timestamp_timezone: '+00:00' }]) { + expect(parsePowerCsvData(NVIDIA_CSV, context).map((row) => row.timestamp)).toEqual([ + '2026-09-11T04:19:41.123Z', + ]); + } + }); + + it('applies a non-zero collector offset', () => { + expect( + parsePowerCsvData(NVIDIA_CSV, { timestamp_timezone: '+08:00' }).map((row) => row.timestamp), + ).toEqual(['2026-09-10T20:19:41.123Z']); + }); + + it('leaves ISO timestamps untouched', () => { + const iso = NVIDIA_CSV.replace('2026/09/11 04:19:41.123', '2026-09-11T04:19:41.123Z'); + expect(parsePowerCsvData(iso, { timestamp_timezone: 'UTC' })[0]?.timestamp).toBe( + '2026-09-11T04:19:41.123Z', + ); + }); +}); diff --git a/packages/app/src/components/gpu-power/power-audit-bundle.ts b/packages/app/src/components/gpu-power/power-audit-bundle.ts new file mode 100644 index 000000000..063972401 --- /dev/null +++ b/packages/app/src/components/gpu-power/power-audit-bundle.ts @@ -0,0 +1,373 @@ +/** + * Cuts per-config power series out of a `power_audit_` bundle. + * + * Slurm / Dynamo disaggregated runners do not publish one `gpu_metrics_*` CSV + * per config. Their DCGM collector writes one bundle per concurrency sweep: + * `LOGS/power/samples.csv` holds ~1 Hz watts for every GPU across the whole + * sweep, `LOGS/power/manifest.json` names the devices and their worker role, + * and one top-level `power_validation_*.json` per concurrency records the + * validated measurement window. Each validation file becomes one + * `GpuPowerSeries` whose `source` is that file's basename, so the timeline + * joins it to the chart row carrying the same `power_audit.source`. + * + * Pure: the route hands over the extracted entry texts; nothing here reads + * the network or the zip. + */ +import { parseNvidiaTimestamp } from '@semianalysisai/inferencex-db/etl/gpu-metrics-csv'; +import { + isPowerAuditValidationEntry, + normalizePowerAuditValidations, +} from '@semianalysisai/inferencex-db/etl/power-audit-validations'; + +import { + bucketPowerSeries, + parseTelemetryTimestampUtc, + type GpuPowerDevice, + type GpuPowerRole, + type GpuPowerSeries, +} from './power-series'; +import { parseCsvData, type GpuMetricRow } from './types'; + +/** Seconds of telemetry kept on each side of a validated window (ramp-up / drain context). */ +export const BUNDLE_WINDOW_PAD_SECONDS = 60; + +export const BUNDLE_SAMPLES_ENTRY = 'LOGS/power/samples.csv'; +export const BUNDLE_MANIFEST_ENTRY = 'LOGS/power/manifest.json'; + +/** Zip entries the route must extract for `cutPowerAuditBundle`; everything else stays compressed. */ +export function isPowerAuditBundleEntry(entryName: string): boolean { + return ( + entryName === BUNDLE_SAMPLES_ENTRY || + entryName === BUNDLE_MANIFEST_ENTRY || + isPowerAuditValidationEntry(entryName) || + isSmiCsv(entryName) || + isContextEntry(entryName) + ); +} + +function isSmiCsv(name: string): boolean { + return /(?:^|\/)gpu_metrics[^/]*\.csv$/u.test(name) && !/_(?:identity|energy_)/u.test(name); +} + +function isContextEntry(entryName: string): boolean { + const name = basename(entryName); + return name.includes('gpu_metrics') && name.toLowerCase().endsWith('_context.json'); +} + +export interface PowerAuditSample { + /** Unix seconds. */ + time: number; + deviceId: string; + power: number; +} + +export interface PowerAuditDevice { + /** `/`, the form `power_audit.observed_gpu_ids` uses. */ + id: string; + hostname: string; + gpuIndex: number; +} + +interface ParsedSamples { + samples: PowerAuditSample[]; + /** Devices in first-seen order. */ + devices: Map; +} + +const SAMPLE_COLUMNS = ['timestamp_unix', 'hostname', 'gpu_index', 'gpu_uuid', 'power_w'] as const; + +const NUMERIC_CELL = /^-?\d+(?:\.\d+)?(?:e[+-]?\d+)?$/iu; + +/** A CSV cell as a number, or NaN for anything that is not a plain numeral (`Number('')` is 0). */ +function numericCell(cell: string): number { + const text = cell.trim(); + return NUMERIC_CELL.test(text) ? Number(text) : Number.NaN; +} + +function splitCsvLines(text: string): string[] { + return text.split('\n').map((line) => (line.endsWith('\r') ? line.slice(0, -1) : line)); +} + +/** Header-driven parse of `samples.csv`; malformed rows are skipped. */ +function parseSamples(csv: string): ParsedSamples { + const lines = splitCsvLines(csv); + const headerIndex = lines.findIndex((line) => line.trim().length > 0); + const empty: ParsedSamples = { samples: [], devices: new Map() }; + if (headerIndex === -1) return empty; + const header = lines[headerIndex].split(',').map((cell) => cell.trim()); + const columns = SAMPLE_COLUMNS.map((name) => header.indexOf(name)); + if (columns.includes(-1)) return empty; + const [timeCol, hostCol, indexCol, uuidCol, powerCol] = columns; + const width = Math.max(...columns) + 1; + + const samples: PowerAuditSample[] = []; + const devices = new Map(); + for (const line of lines.slice(headerIndex + 1)) { + const cells = line.split(','); + if (cells.length < width) continue; + const time = numericCell(cells[timeCol]); + const power = numericCell(cells[powerCol]); + const gpuIndex = numericCell(cells[indexCol]); + const hostname = cells[hostCol].trim(); + const uuid = cells[uuidCol].trim(); + if ( + !Number.isFinite(time) || + !Number.isFinite(power) || + !Number.isInteger(gpuIndex) || + hostname.length === 0 || + uuid.length === 0 + ) { + continue; + } + const deviceId = `${hostname}/${uuid}`; + if (!devices.has(deviceId)) devices.set(deviceId, { id: deviceId, hostname, gpuIndex }); + samples.push({ time, deviceId, power }); + } + return { samples, devices }; +} + +function asRole(value: unknown): GpuPowerRole | undefined { + return value === 'prefill' || value === 'decode' ? value : undefined; +} + +function parseJsonObject(text: string | undefined): Record | null { + if (text === undefined) return null; + try { + const parsed: unknown = JSON.parse(text); + return parsed !== null && typeof parsed === 'object' && !Array.isArray(parsed) + ? (parsed as Record) + : null; + } catch { + return null; + } +} + +function isRecord(value: unknown): value is Record { + return value !== null && typeof value === 'object' && !Array.isArray(value); +} + +function manifestSlot(hostname: string, gpuIndex: number): string { + return `${hostname}#${gpuIndex}`; +} + +/** `expected_devices[].assignments[0].worker_role`, keyed by hostname + gpu_index. */ +function manifestRoles(manifest: Record | null): Map { + const roles = new Map(); + const expected = manifest?.expected_devices; + if (!Array.isArray(expected)) return roles; + for (const device of expected) { + if (!isRecord(device)) continue; + const { hostname, gpu_index: gpuIndex, assignments } = device; + if ( + typeof hostname !== 'string' || + typeof gpuIndex !== 'number' || + !Array.isArray(assignments) + ) { + continue; + } + const first: unknown = assignments[0]; + const role = isRecord(first) ? asRole(first.worker_role) : undefined; + if (role) roles.set(manifestSlot(hostname, gpuIndex), role); + } + return roles; +} + +interface TimeWindow { + start: number; + end: number; +} + +/** `selected_window`, else `benchmark_window`; `null` when neither has a finite range. */ +function validationWindow(validation: Record): TimeWindow | null { + for (const key of ['selected_window', 'benchmark_window']) { + const window = validation[key]; + if (!isRecord(window)) continue; + const start = window.start_time_unix; + const end = window.end_time_unix; + if ( + typeof start === 'number' && + typeof end === 'number' && + Number.isFinite(start) && + Number.isFinite(end) && + end >= start + ) { + return { start, end }; + } + } + return null; +} + +const ROLE_ORDER: Record = { prefill: 0, decode: 1 }; + +function roleRank(role: GpuPowerRole | undefined): number { + return role ? ROLE_ORDER[role] : 2; +} + +function compareText(a: string, b: string): number { + if (a < b) return -1; + return a > b ? 1 : 0; +} + +/** Prefill, then decode, then unassigned; within a role by hostname, then gpu_index. */ +function orderDevices( + devices: readonly GpuPowerDevice[], + byId: ReadonlyMap, +) { + return devices.toSorted((a, b) => { + const rank = roleRank(a.role) - roleRank(b.role); + if (rank !== 0) return rank; + const deviceA = byId.get(a.id)!; + const deviceB = byId.get(b.id)!; + return compareText(deviceA.hostname, deviceB.hostname) || deviceA.gpuIndex - deviceB.gpuIndex; + }); +} + +function basename(entryName: string): string { + return entryName.slice(entryName.lastIndexOf('/') + 1); +} + +/** + * Normalize NVIDIA's unzoned wall-clock timestamps to ISO UTC, matching the ingest + * parser: the adjacent collector context supplies the offset and missing or UTC + * context means zero. Live and stored reads then agree, so a run does not shift by + * the browser timezone before ingest. ISO and AMD timestamps pass through unchanged. + */ +export function parsePowerCsvData( + text: string, + context: Record | null, +): GpuMetricRow[] { + const zone = context?.timestamp_timezone; + const offset = + typeof zone === 'string' + ? /^(?[+-])(?\d{2}):?(?\d{2})$/u.exec(zone.trim())?.groups + : null; + const offsetMinutes = offset + ? (offset.sign === '-' ? -1 : 1) * (Number(offset.h) * 60 + Number(offset.m)) + : 0; + return parseCsvData(text).map((row) => { + const timestamp = parseNvidiaTimestamp(row.timestamp, offsetMinutes); + return timestamp === null ? row : { ...row, timestamp: new Date(timestamp).toISOString() }; + }); +} + +/** + * One series per validation file that has a window and samples within + * `BUNDLE_WINDOW_PAD_SECONDS` of it. Rows follow `devices`, ordered prefill, + * decode, unassigned (hostname, then gpu_index inside a role); `gpus` are the + * row indices `0..n-1`. Roles come from the validation's `per_gpu_role`, else + * the manifest's `expected_devices`. Malformed JSON skips that file; a + * missing or empty `samples.csv` yields `[]`. Output is ordered by window start. + */ +export function cutPowerAuditBundle( + artifact: string, + files: ReadonlyMap, +): GpuPowerSeries[] { + const validations = normalizePowerAuditValidations(artifact, files); + const manifest = parseJsonObject(files.get(BUNDLE_MANIFEST_ENTRY)); + const contextFiles = [...files] + .filter(([name]) => isContextEntry(name)) + .sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)); + const smiFiles = [...files] + .filter(([name]) => isSmiCsv(name)) + .map(([name, text]) => { + const directory = name.slice(0, name.lastIndexOf('/') + 1); + let context: Record | null = null; + for (const [entry, contents] of contextFiles) { + if (entry.slice(0, entry.lastIndexOf('/') + 1) !== directory) continue; + context = parseJsonObject(contents); + if (context) break; + } + const data = parsePowerCsvData(text, context); + return { name, data }; + }) + .filter((file) => file.data.length > 0); + // Ingest prefers the richer SMI CSV when the bundle carries both collectors. + if (smiFiles.length > 0) return cutPowerAuditCsvs(artifact, smiFiles, validations, manifest); + const samplesText = files.get(BUNDLE_SAMPLES_ENTRY); + if (samplesText === undefined) return []; + const { samples, devices } = parseSamples(samplesText); + if (samples.length === 0) return []; + return cutPowerAuditSamples( + artifact, + samples, + devices, + validations, + parseJsonObject(files.get(BUNDLE_MANIFEST_ENTRY)), + ); +} + +/** Apply bundle window cuts to the richer SMI CSVs without requiring DCGM UUID sidecars. */ +export function cutPowerAuditCsvs( + artifact: string, + files: readonly { name: string; data: readonly GpuMetricRow[] }[], + validations: ReadonlyMap>, + manifest: Record | null, +): GpuPowerSeries[] { + const populated = files.filter((file) => file.data.length > 0); + const devices = new Map(); + const samples: PowerAuditSample[] = []; + for (const file of populated) { + for (const row of file.data) { + const id = populated.length === 1 ? String(row.index) : `${file.name}/${row.index}`; + devices.set(id, { id, hostname: file.name, gpuIndex: row.index }); + const time = parseTelemetryTimestampUtc(row.timestamp); + if (time !== null) samples.push({ deviceId: id, time: time / 1000, power: row.power }); + } + } + return cutPowerAuditSamples(artifact, samples, devices, validations, manifest); +} + +/** The same window/device/bucketing transform for persisted samples and artifact CSVs. */ +export function cutPowerAuditSamples( + artifact: string, + samples: readonly PowerAuditSample[], + devices: ReadonlyMap, + validations: ReadonlyMap>, + manifest: Record | null, +): GpuPowerSeries[] { + const fallbackRoles = manifestRoles(manifest); + const cut: { start: number; series: GpuPowerSeries }[] = []; + for (const [entryName, validation] of validations) { + const window = validationWindow(validation); + if (!window) continue; + + const lo = window.start - BUNDLE_WINDOW_PAD_SECONDS; + const hi = window.end + BUNDLE_WINDOW_PAD_SECONDS; + const inRange = samples.filter((sample) => sample.time >= lo && sample.time <= hi); + if (inRange.length === 0) continue; + + const explicitRoles = isRecord(validation.per_gpu_role) ? validation.per_gpu_role : {}; + const present = [...new Set(inRange.map((sample) => sample.deviceId))].map((id) => { + const device = devices.get(id)!; + const role = + asRole(explicitRoles[id]) ?? + fallbackRoles.get(manifestSlot(device.hostname, device.gpuIndex)); + return role ? { id, role } : { id }; + }); + const ordered = orderDevices(present, devices); + const rowOf = new Map(ordered.map((device, row) => [device.id, row])); + + const rows: GpuMetricRow[] = inRange.map((sample) => ({ + timestamp: String(sample.time), + index: rowOf.get(sample.deviceId)!, + power: sample.power, + })); + const series = bucketPowerSeries(artifact, rows); + if (!series) continue; + // The synthetic `index` is the device's position in `ordered`, so each + // bucketed row maps back to its device; `gpus` becomes plain row indices. + cut.push({ + start: window.start, + series: { + ...series, + artifact, + source: basename(entryName), + gpus: series.gpus.map((_, row) => row), + devices: series.gpus.map((row) => ordered[row]), + }, + }); + } + return cut + .toSorted((a, b) => a.start - b.start || compareText(a.series.source!, b.series.source!)) + .map((entry) => entry.series); +} diff --git a/packages/app/src/components/gpu-power/power-series.test.ts b/packages/app/src/components/gpu-power/power-series.test.ts new file mode 100644 index 000000000..4b5a0fd91 --- /dev/null +++ b/packages/app/src/components/gpu-power/power-series.test.ts @@ -0,0 +1,103 @@ +import { describe, expect, it } from 'vitest'; + +import { + bucketPowerSeries, + bucketPowerFiles, + bucketTimeMs, + meanPowerAt, + sumPowerAt, + type GpuPowerSeries, +} from './power-series'; +import type { GpuMetricRow } from './types'; + +function row(timestamp: string, index: number, power: number): GpuMetricRow { + return { + timestamp, + index, + power, + temperature: 40, + smClock: 1, + memClock: 1, + gpuUtil: 0, + memUtil: 0, + }; +} + +describe('bucketPowerSeries', () => { + it('aligns every GPU on one-second buckets and marks missing samples null', () => { + const rows = [ + row('2026/09/12 20:20:41.817', 0, 194.4), + row('2026/09/12 20:20:41.820', 1, 192.4), + row('2026/09/12 20:20:42.816', 0, 300), + // GPU 1 skipped this tick; GPU 0 sampled twice in the next one. + row('2026/09/12 20:20:43.815', 0, 400), + row('2026/09/12 20:20:43.900', 0, 500), + row('2026/09/12 20:20:43.818', 1, 350), + ]; + const series = bucketPowerSeries('gpu_metrics_test', rows); + expect(series).not.toBeNull(); + expect(series!.startMs).toBe(Date.UTC(2026, 8, 12, 20, 20, 41)); + expect(series!.bucketSeconds).toBe(1); + expect(series!.gpus).toEqual([0, 1]); + expect(series!.t).toEqual([0, 1, 2]); + expect(series!.power).toEqual([ + [194.4, 300, 450], + [192.4, null, 350], + ]); + expect(meanPowerAt(series!, 1)).toBe(300); + expect(meanPowerAt(series!, 2)).toBe(400); + expect(bucketTimeMs(series!, 2)).toBe(Date.UTC(2026, 8, 12, 20, 20, 43)); + }); +}); + +describe('sumPowerAt', () => { + const series: GpuPowerSeries = { + artifact: 'power_audit_sweep', + startMs: Date.UTC(2026, 8, 12, 20, 0, 0), + bucketSeconds: 1, + gpus: [0, 1, 2], + t: [0, 1, 2], + power: [ + [100, null, 300], + [50, 60, null], + [10, null, null], + ], + }; + + it('returns null for a partial pool, an unsampled bucket, an empty pool or an unknown row', () => { + // GPU 0 and 2 have no sample in bucket 1: a partial sum would read as a dip. + expect(sumPowerAt(series, [0, 1, 2], 1)).toBeNull(); + expect(sumPowerAt(series, [0, 1], 2)).toBeNull(); + expect(sumPowerAt(series, [0, 2], 1)).toBeNull(); + expect(sumPowerAt(series, [], 0)).toBeNull(); + expect(sumPowerAt(series, [7], 0)).toBeNull(); + }); +}); + +describe('bucketPowerFiles', () => { + it('keeps repeated GPU indices from different host files separate with stable identity', () => { + const time = '2026-09-12T04:00:00Z'; + const files = [ + { name: 'host-b/gpu_metrics.csv', data: [row(time, 0, 500)] }, + { name: 'host-a/gpu_metrics.csv', data: [row(time, 0, 100), row(time, 2, 200)] }, + { name: 'empty.csv', data: [] }, + ]; + const expected = { + artifact: 'gpu_metrics_multinode', + startMs: Date.parse(time), + bucketSeconds: 1, + gpus: [0, 1, 2], + t: [0], + power: [[100], [200], [500]], + devices: [ + { id: 'host-a/gpu_metrics.csv/0' }, + { id: 'host-a/gpu_metrics.csv/2' }, + { id: 'host-b/gpu_metrics.csv/0' }, + ], + }; + expect(bucketPowerFiles('gpu_metrics_multinode', files)).toEqual(expected); + expect(bucketPowerFiles('gpu_metrics_multinode', files.toReversed())).toEqual(expected); + expect(bucketPowerFiles('empty', [])).toBeNull(); + expect(bucketPowerFiles('single', [files[0]])?.gpus).toEqual([0]); + }); +}); diff --git a/packages/app/src/components/gpu-power/power-series.ts b/packages/app/src/components/gpu-power/power-series.ts new file mode 100644 index 000000000..8ff75fabb --- /dev/null +++ b/packages/app/src/components/gpu-power/power-series.ts @@ -0,0 +1,212 @@ +/** + * Compact per-GPU power series for the PowerX timeline. + * + * A `gpu_metrics_` artifact carries ~1 Hz samples for every + * GPU of one benchmark config (server start, warmup, and the measured window). + * The raw rows are heavy (a 25-config run is ~27 MB of JSON), so the API can + * return this columnar shape instead: one row of watts per GPU, aligned on + * fixed-width time buckets, with `null` where a GPU had no sample. + * + * Runner timestamps are UTC wall clock (`2026/09/12 20:19:57.158`); the + * validated measurement window in `power_audit` is UTC epoch seconds, so the + * two line up without a timezone guess. + */ +import type { GpuMetricRow, GpuPowerRunInfo } from './types'; + +/** Worker role a GPU served, when the collector's manifest assigns one. */ +export type GpuPowerRole = 'prefill' | 'decode'; + +export interface GpuPowerDevice { + /** + * Producer device identifier: the nvidia-smi / amd-smi index (`"3"`) for + * `gpu_metrics_*` CSVs, `/` for DCGM power-audit + * bundles — the same form as `power_audit.observed_gpu_ids` on the row. + */ + id: string; + role?: GpuPowerRole; +} + +export interface GpuPowerSeries { + /** GitHub artifact name: `gpu_metrics_` or `power_audit_`. */ + artifact: string; + /** + * Basename of the `power_validation_*.json` this series belongs to, for + * series cut from a power-audit bundle (one bundle holds a whole + * concurrency sweep). Absent for `gpu_metrics_*` series, whose artifact + * already names one config. + */ + source?: string; + /** UTC epoch milliseconds of bucket 0. */ + startMs: number; + /** Bucket width in seconds. */ + bucketSeconds: number; + /** GPU indices, one per row of `power`, ascending. */ + gpus: number[]; + /** Seconds since `startMs` for each bucket, ascending. */ + t: number[]; + /** Watts per GPU (outer) per bucket (inner); `null` when no sample landed in the bucket. */ + power: (number | null)[][]; + /** One entry per row of `power` when the collector reports device identity or roles. */ + devices?: GpuPowerDevice[]; +} + +export interface GpuPowerSeriesResponse { + runInfo: GpuPowerRunInfo; + series: GpuPowerSeries[]; + /** Coverage of requested validation identities only, never the full run/sweep. */ + sourceCoverage?: { + status: 'complete' | 'incomplete' | 'unknown'; + missingSources: string[]; + }; +} + +const RUNNER_TIMESTAMP = + /^(?\d{4})[/-](?\d{2})[/-](?\d{2})[ T](?\d{2}):(?\d{2}):(?\d{2})(?:\.(?\d{1,6}))?$/u; + +/** + * Parses a runner telemetry timestamp as UTC epoch milliseconds. + * + * nvidia-smi / amd-smi write `YYYY/MM/DD HH:MM:SS.mmm` without a zone; the + * collectors run in UTC, and `new Date(...)` would read that as browser local + * time, so the digits are assembled explicitly. ISO strings with a zone and + * numeric epochs (seconds or milliseconds) are accepted as-is. + */ +export function parseTelemetryTimestampUtc(raw: string): number | null { + const text = raw.trim(); + const match = RUNNER_TIMESTAMP.exec(text); + if (match?.groups) { + const { y, mo, d, h, mi, s, ms } = match.groups; + const millis = ms ? Math.round(Number(`0.${ms}`) * 1000) : 0; + return Date.UTC(Number(y), Number(mo) - 1, Number(d), Number(h), Number(mi), Number(s), millis); + } + if (/^\d+(?:\.\d+)?$/u.test(text)) { + const numeric = Number(text); + return numeric < 1e12 ? Math.round(numeric * 1000) : Math.round(numeric); + } + const parsed = Date.parse(text); + return Number.isNaN(parsed) ? null : parsed; +} + +/** + * Buckets raw per-GPU rows into the compact series. Samples of one GPU that + * share a bucket are averaged; repeated GPU/timestamps keep the first sample, + * matching ingest. Buckets nobody sampled are omitted from `t`. + * Returns `null` when no row carries a parseable timestamp. + */ +export function bucketPowerSeries( + artifact: string, + rows: readonly GpuMetricRow[], + bucketSeconds = 1, +): GpuPowerSeries | null { + if (!(bucketSeconds > 0)) throw new RangeError('bucketSeconds must be positive'); + const bucketMs = bucketSeconds * 1000; + let startMs = Number.POSITIVE_INFINITY; + const parsed: { ms: number; gpu: number; power: number }[] = []; + const seen = new Set(); + for (const row of rows) { + if (!Number.isFinite(row.power) || !Number.isInteger(row.index)) continue; + const ms = parseTelemetryTimestampUtc(row.timestamp); + if (ms === null) continue; + const key = `${row.index}:${ms}`; + if (seen.has(key)) continue; + seen.add(key); + parsed.push({ ms, gpu: row.index, power: row.power }); + if (ms < startMs) startMs = ms; + } + if (parsed.length === 0) return null; + startMs = Math.floor(startMs / bucketMs) * bucketMs; + + const gpus = [...new Set(parsed.map((sample) => sample.gpu))].toSorted((a, b) => a - b); + const gpuRow = new Map(gpus.map((gpu, row) => [gpu, row])); + // bucket index -> per GPU [sum, count] + const buckets = new Map(); + for (const sample of parsed) { + const bucket = Math.floor((sample.ms - startMs) / bucketMs); + let cells = buckets.get(bucket); + if (!cells) { + cells = gpus.map(() => [0, 0]); + buckets.set(bucket, cells); + } + const cell = cells[gpuRow.get(sample.gpu)!]; + cell[0] += sample.power; + cell[1] += 1; + } + const bucketIndices = [...buckets.keys()].toSorted((a, b) => a - b); + const power: (number | null)[][] = gpus.map(() => + Array.from({ length: bucketIndices.length }, (): number | null => null), + ); + bucketIndices.forEach((bucket, column) => { + const cells = buckets.get(bucket)!; + cells.forEach(([sum, count], row) => { + power[row][column] = count > 0 ? Math.round((sum / count) * 100) / 100 : null; + }); + }); + return { + artifact, + startMs, + bucketSeconds, + gpus, + t: bucketIndices.map((bucket) => bucket * bucketSeconds), + power, + }; +} + +/** Keep host-local GPU indices distinct when one artifact stages multiple CSVs. */ +export function bucketPowerFiles( + artifact: string, + files: readonly { name: string; data: readonly GpuMetricRow[] }[], +): GpuPowerSeries | null { + const populated = files + .filter((file) => file.data.length > 0) + .toSorted((a, b) => a.name.localeCompare(b.name)); + if (populated.length === 0) return null; + if (populated.length === 1) return bucketPowerSeries(artifact, populated[0].data); + const devices = populated.flatMap((file) => + [...new Set(file.data.map((row) => row.index))] + .toSorted((a, b) => a - b) + .map((gpu) => ({ id: `${file.name}/${gpu}`, file, gpu })), + ); + const rows = devices.flatMap(({ file, gpu }, index) => + file.data.filter((row) => row.index === gpu).map((row) => ({ ...row, index })), + ); + const bucketed = bucketPowerSeries(artifact, rows); + return bucketed ? { ...bucketed, devices: devices.map(({ id }) => ({ id })) } : null; +} + +/** Mean watts across the GPUs that have a sample in bucket `column`, or `null`. */ +export function meanPowerAt(series: GpuPowerSeries, column: number): number | null { + let sum = 0; + let count = 0; + for (const row of series.power) { + const value = row[column]; + if (value === null || value === undefined) continue; + sum += value; + count += 1; + } + return count > 0 ? sum / count : null; +} + +/** + * Summed watts over `rows` in bucket `column`, or `null` unless EVERY row has + * a sample: a pool total with a device missing would read as a dip against + * the pool's TDP, so the bucket is left as a gap instead. + */ +export function sumPowerAt( + series: GpuPowerSeries, + rows: readonly number[], + column: number, +): number | null { + if (rows.length === 0) return null; + let sum = 0; + for (const row of rows) { + const value = series.power[row]?.[column]; + if (value === null || value === undefined) return null; + sum += value; + } + return sum; +} + +/** UTC epoch milliseconds of bucket `column`. */ +export function bucketTimeMs(series: GpuPowerSeries, column: number): number { + return series.startMs + series.t[column] * 1000; +} diff --git a/packages/app/src/components/gpu-power/stored-gpu-stats.test.ts b/packages/app/src/components/gpu-power/stored-gpu-stats.test.ts new file mode 100644 index 000000000..893b1ae2a --- /dev/null +++ b/packages/app/src/components/gpu-power/stored-gpu-stats.test.ts @@ -0,0 +1,128 @@ +import { describe, expect, it } from 'vitest'; +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { + computeGpuMetricStats, + parseGpuMetricsCsv, + type GpuMetricSample, +} from '@semianalysisai/inferencex-db/etl/gpu-metrics-csv'; +import { + prepareGpuMetricsArtifact, + statMetricColumn, +} from '@semianalysisai/inferencex-db/etl/gpu-metrics-ingest'; + +import { storedGpuStatsForMetric } from './stored-gpu-stats'; +import { ALL_METRIC_OPTIONS, computeGpuStats, parseCsvData, type GpuMetricRow } from './types'; + +function storedStats(samples: GpuMetricSample[]) { + return computeGpuMetricStats(samples).map((row) => ({ + ...row, + metric: statMetricColumn(row.metric), + })); +} + +function assertAllMetricStats(samples: GpuMetricSample[], rows: GpuMetricRow[]) { + const digest = storedStats(samples); + for (const { key } of ALL_METRIC_OPTIONS) { + const actual = storedGpuStatsForMetric(digest, key); + const expected = computeGpuStats(rows, key); + expect(actual, key).toHaveLength(expected.length); + for (const [index, row] of actual.entries()) { + expect(row.gpuIndex).toBe(expected[index].gpuIndex); + expect(row.count).toBe(expected[index].count); + // ETL sums sorted samples while the live reader sums recording order. + for (const field of ['min', 'max', 'mean', 'median', 'p95', 'p99', 'stddev'] as const) { + expect(row[field], `${key}.${field}`).toBeCloseTo(expected[index][field], 10); + } + } + } +} + +describe('storedGpuStatsForMetric', () => { + it.each(['UTC', 'America/Los_Angeles'])( + 'matches live NVIDIA and ingest populations across DST in %s', + (timezone) => { + const csv = [ + 'timestamp,index,power.draw [W]', + '2026/03/08 02:30:00.000,0,100', + '2026/03/08 02:30:00.000,0,900', + '2026/03/08 03:30:00.000,0,300', + ].join('\n'); + const artifactDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gpu-stats-dst-')); + const originalTimezone = process.env.TZ; + try { + process.env.TZ = timezone; + fs.writeFileSync(path.join(artifactDir, 'gpu_metrics.csv'), csv); + const [prepared] = prepareGpuMetricsArtifact({ + artifactDir, + artifactName: 'gpu_metrics_nvidia_dst', + }); + expect( + prepared.samples.map((sample) => new Date(sample.timestampMs).toISOString()), + ).toEqual(['2026-03-08T02:30:00.000Z', '2026-03-08T03:30:00.000Z']); + const live = parseCsvData(csv); + // Preserve raw timestamps for the bundle's later context-offset adjustment. + expect(live.map((row) => row.timestamp)).toEqual([ + '2026/03/08 02:30:00.000', + '2026/03/08 03:30:00.000', + ]); + expect(live.map((row) => row.power)).toEqual([100, 300]); + const actual = computeGpuStats(live, 'power'); + expect(actual).toEqual([ + { + gpuIndex: 0, + count: 2, + min: 100, + max: 300, + mean: 200, + median: 200, + p95: 290, + p99: 298, + stddev: 100, + }, + ]); + const { metric: _metric, ...expected } = prepared.stats[0]; + expect(actual).toEqual([expected]); + } finally { + if (originalTimezone === undefined) delete process.env.TZ; + else process.env.TZ = originalTimezone; + fs.rmSync(artifactDir, { recursive: true, force: true }); + } + }, + ); + + it('preserves NVIDIA units, zero readings, and full-record percentiles', () => { + const csv = [ + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]', + '2026/09/21 00:00:00, 0, 0 W, N/A, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/21 00:00:01, 0, 912.1 W, 61, 1965 MHz, 3996 MHz, 98 %, 74 %', + '2026/09/21 00:00:02, 0, 187.8 W, 35, 120 MHz, 3996 MHz, 0 %, 0 %', + // Conflicting stop-time flush must not change the first reading. + '2026/09/21 00:00:02, 0, 999 W, 90, 120 MHz, 3996 MHz, 0 %, 0 %', + ].join('\n'); + const samples = parseGpuMetricsCsv(csv)!.samples.slice(0, 3); + assertAllMetricStats(samples, parseCsvData(csv)); + const [power] = storedGpuStatsForMetric(storedStats(samples), 'power'); + expect(power.count).toBe(3); + expect(power.median).toBe(187.8); + expect(power.p95).toBeCloseTo(839.67, 10); + expect(power.p99).toBeCloseTo(897.614, 10); + }); + + it('maps every AMD metric without converting missing readings to zero', () => { + const csv = [ + 'timestamp,gpu,gfx_activity,umc_activity,mm_activity,socket_power,gfx_voltage,soc_voltage,mem_voltage,gfx_0_clk,mem_0_clk,fclk_0_clk,socclk_0_clk,edge,hotspot,mem', + '1789948800,0,0,0,N/A,0,N/A,N/A,N/A,N/A,2000,1250,39,N/A,N/A,24', + '1789948801,0,95,80,5,1180,850,900,1250,2402,2050,1300,40,55,78,60', + '1789948801,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0', + ].join('\n'); + const samples = parseGpuMetricsCsv(csv)!.samples.slice(0, 2); + assertAllMetricStats(samples, parseCsvData(csv)); + expect(storedGpuStatsForMetric(storedStats(samples), 'gfxVoltage')[0]).toMatchObject({ + count: 1, + mean: 850, + stddev: 0, + }); + }); +}); diff --git a/packages/app/src/components/gpu-power/stored-gpu-stats.ts b/packages/app/src/components/gpu-power/stored-gpu-stats.ts new file mode 100644 index 000000000..5eda4d3b2 --- /dev/null +++ b/packages/app/src/components/gpu-power/stored-gpu-stats.ts @@ -0,0 +1,30 @@ +import type { GpuMetricStatRow } from '@semianalysisai/inferencex-db/queries/gpu-metrics'; + +import type { GpuMetricKey, GpuStats } from './types'; + +const STORED_METRIC: Record = { + power: 'power_w', + temperature: 'temperature_c', + smClock: 'sm_clock_mhz', + memClock: 'mem_clock_mhz', + gpuUtil: 'gpu_util_pct', + memUtil: 'mem_util_pct', + edgeTemp: 'edge_temp_c', + memTemp: 'mem_temp_c', + gfxVoltage: 'gfx_voltage_mv', + socVoltage: 'soc_voltage_mv', + memVoltage: 'mem_voltage_mv', + fclk: 'fclk_mhz', + socClk: 'socclk_mhz', + mmActivity: 'mm_activity_pct', +}; + +/** Full-record statistics, including startup/warmup; never a measured-window summary. */ +export function storedGpuStatsForMetric( + stats: readonly GpuMetricStatRow[], + metricKey: GpuMetricKey, +): GpuStats[] { + return stats + .filter((row) => row.metric === STORED_METRIC[metricKey]) + .map(({ metric: _metric, ...row }) => row); +} diff --git a/packages/app/src/components/gpu-power/stored-power-series.ts b/packages/app/src/components/gpu-power/stored-power-series.ts new file mode 100644 index 000000000..964d44812 --- /dev/null +++ b/packages/app/src/components/gpu-power/stored-power-series.ts @@ -0,0 +1,168 @@ +import type { GpuMetricSeries } from '@semianalysisai/inferencex-db/queries/gpu-metrics'; + +import { + cutPowerAuditSamples, + cutPowerAuditCsvs, + type PowerAuditDevice, + type PowerAuditSample, +} from './power-audit-bundle'; +import { bucketPowerFiles, type GpuPowerSeries } from './power-series'; + +function isRecord(value: unknown): value is Record { + return value !== null && typeof value === 'object' && !Array.isArray(value); +} + +export class StoredTelemetryIncompleteError extends Error { + readonly artifact: string; + + constructor(artifact: string, reason: string) { + super(`${artifact}: ${reason}. Re-ingest this artifact for the same run and attempt.`); + this.name = 'StoredTelemetryIncompleteError'; + this.artifact = artifact; + } +} + +/** Compare storage against the original parsed artifact, not only surviving host rows. */ +function checkSeriesInventory(artifact: string, series: readonly GpuMetricSeries[]): void { + for (const entry of series) { + const inventory = entry.sidecars.seriesInventory; + if (!Array.isArray(inventory)) continue; + for (const expected of inventory) { + if ( + !isRecord(expected) || + typeof expected.fileName !== 'string' || + typeof expected.sampleCount !== 'number' + ) + continue; + const actual = series.find((candidate) => candidate.fileName === expected.fileName); + if (!actual || actual.data.length !== expected.sampleCount) { + throw new StoredTelemetryIncompleteError( + artifact, + `stored file/sample coverage is incomplete for ${expected.fileName}`, + ); + } + } + } +} + +/** Reconstruct only source/window evidence retained by the pre-validation ingest. */ +function storedValidations(series: readonly GpuMetricSeries[]) { + const validations = new Map>(); + for (const entry of series) { + if (!isRecord(entry.sidecars.validations)) continue; + for (const [name, validation] of Object.entries(entry.sidecars.validations)) { + if (/^power_validation_[^/]+\.json$/u.test(name) && isRecord(validation)) { + validations.set(name, validation); + } + } + } + // New ingests carry the complete original documents, including role overrides. + // The legacy path has source + the selected serving window and manifest roles. + if (validations.size > 0) return validations; + for (const entry of series) { + for (const audit of entry.powerAudits ?? []) { + const { source, window_start_unix: start, window_end_unix: end } = audit; + if ( + typeof source !== 'string' || + typeof start !== 'number' || + typeof end !== 'number' || + !Number.isFinite(start) || + !Number.isFinite(end) || + end < start + ) + continue; + const name = source.slice(source.lastIndexOf('/') + 1); + if (!/^power_validation_[^/]+\.json$/u.test(name)) continue; + validations.set(name, { selected_window: { start_time_unix: start, end_time_unix: end } }); + } + } + return validations; +} + +function storedBundle(artifact: string, series: readonly GpuMetricSeries[]): GpuPowerSeries[] { + const validations = storedValidations(series); + if (validations.size === 0) + throw new StoredTelemetryIncompleteError(artifact, 'validation window provenance is missing'); + const smiFiles = series.filter((entry) => /(?:^|\/)gpu_metrics[^/]*\.csv$/u.test(entry.fileName)); + if (smiFiles.length > 0) { + const manifest = smiFiles.find((entry) => isRecord(entry.sidecars.powerManifest))?.sidecars + .powerManifest; + return cutPowerAuditCsvs( + artifact, + smiFiles.map((entry) => ({ name: entry.fileName, data: entry.data })), + validations, + isRecord(manifest) ? manifest : null, + ); + } + const devices = new Map(); + const samples: PowerAuditSample[] = []; + for (const entry of series) { + const identity = entry.sidecars.identity; + if (!Array.isArray(identity)) continue; + const ids = new Map(); + for (const item of identity) { + if (!isRecord(item)) continue; + const { hostname, gpu_index: gpuIndex, gpu_uuid: uuid } = item; + if (typeof hostname !== 'string' || typeof gpuIndex !== 'number' || typeof uuid !== 'string') + continue; + const id = `${hostname}/${uuid}`; + ids.set(gpuIndex, id); + devices.set(id, { id, hostname, gpuIndex }); + } + for (const row of entry.data) { + const deviceId = ids.get(row.index); + if (!deviceId) + throw new StoredTelemetryIncompleteError( + artifact, + `device identity is missing for ${entry.fileName} GPU ${row.index}`, + ); + samples.push({ deviceId, time: Date.parse(row.timestamp) / 1000, power: row.power }); + } + } + const manifest = series.find((entry) => isRecord(entry.sidecars.context))?.sidecars.context; + const expectedDevices = isRecord(manifest) ? manifest.expected_devices : null; + if (Array.isArray(expectedDevices)) { + const slots = new Set( + [...devices.values()].map((device) => `${device.hostname}#${device.gpuIndex}`), + ); + for (const expected of expectedDevices) { + if ( + isRecord(expected) && + typeof expected.hostname === 'string' && + typeof expected.gpu_index === 'number' && + !slots.has(`${expected.hostname}#${expected.gpu_index}`) + ) { + throw new StoredTelemetryIncompleteError( + artifact, + `expected device ${expected.hostname} GPU ${expected.gpu_index} is not stored`, + ); + } + } + } + return cutPowerAuditSamples( + artifact, + samples, + devices, + validations, + isRecord(manifest) ? manifest : null, + ); +} + +/** Apply the artifact path's existing cuts and one-second means to persisted rows. */ +export function storedPowerSeries(series: readonly GpuMetricSeries[]): GpuPowerSeries[] { + const groups = new Map(); + for (const entry of series) { + const group = groups.get(entry.artifactName) ?? []; + group.push(entry); + groups.set(entry.artifactName, group); + } + return [...groups].flatMap(([artifact, entries]) => { + checkSeriesInventory(artifact, entries); + if (artifact.startsWith('power_audit_')) return storedBundle(artifact, entries); + const bucketed = bucketPowerFiles( + artifact, + entries.map((entry) => ({ name: entry.fileName, data: entry.data })), + ); + return bucketed ? [bucketed] : []; + }); +} diff --git a/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts b/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts new file mode 100644 index 000000000..c181eac2d --- /dev/null +++ b/packages/app/src/components/gpu-power/telemetry-smoothing.test.ts @@ -0,0 +1,34 @@ +import { describe, expect, it } from 'vitest'; + +import { meanAcrossSeries, rollingTimeAverage, type TimedSample } from './telemetry-smoothing'; + +const T0 = Date.parse('2026-09-15T21:36:14.107Z'); + +/** One sample per second starting at T0, values from `values`. */ +function everySecond(values: number[], offsetMs = 0): TimedSample[] { + return values.map((value, i) => ({ ms: T0 + offsetMs + i * 1000, value })); +} + +describe('rollingTimeAverage', () => { + it('averages a centered window and shrinks it at the series edges', () => { + // 1 s cadence, 2 s window => each sample sees itself and its immediate neighbours. + const out = rollingTimeAverage(everySecond([100, 200, 600, 200, 100]), 2000); + expect(out.map((p) => p.value)).toEqual([150, 300, 1000 / 3, 300, 150]); + expect(out.map((p) => p.ms)).toEqual(everySecond([0, 0, 0, 0, 0]).map((p) => p.ms)); + }); +}); + +describe('meanAcrossSeries', () => { + it('picks the nearest sample when a chip dropped a row', () => { + const chip0 = everySecond([0, 0, 0, 0]); + const chip1: TimedSample[] = [ + { ms: T0, value: 10 }, + // row at T0 + 1000 dropped + { ms: T0 + 2000, value: 30 }, + { ms: T0 + 3000, value: 40 }, + ]; + const out = meanAcrossSeries([chip0, chip1], 400); + expect(out.map((p) => p.count)).toEqual([2, 1, 2, 2]); + expect(out.map((p) => p.value)).toEqual([5, 0, 15, 20]); + }); +}); diff --git a/packages/app/src/components/gpu-power/telemetry-smoothing.ts b/packages/app/src/components/gpu-power/telemetry-smoothing.ts new file mode 100644 index 000000000..a9ec9a307 --- /dev/null +++ b/packages/app/src/components/gpu-power/telemetry-smoothing.ts @@ -0,0 +1,160 @@ +/** + * Pure time-series helpers for the PowerX telemetry charts. They operate on + * absolute millisecond timestamps so irregular sampling (dropped rows, a few + * ms of skew between chips) is handled by time, not by sample index. + */ + +export interface TimedSample { + /** Absolute timestamp in milliseconds since the Unix epoch. */ + ms: number; + value: number; +} + +/** Rolling-average window choices offered by the chart controls, in seconds. */ +export const SMOOTHING_WINDOWS_S = [10, 30, 60, 300] as const; +export type SmoothingWindowS = (typeof SMOOTHING_WINDOWS_S)[number]; + +export interface AggregatedSample extends TimedSample { + /** Number of chips that contributed a sample within tolerance. */ + count: number; +} + +export type TelemetryDisplayMode = 'points' | 'rolling'; +/** Per-chip lines, one mean line across the visible chips, or both. */ +export type TelemetrySeriesMode = 'chips' | 'mean' | 'both'; + +export interface TelemetryDisplayState { + mode: TelemetryDisplayMode; + windowS: SmoothingWindowS; + series: TelemetrySeriesMode; +} + +/** Rolling average is the default: at 1 s cadence the raw points read as noise. */ +export const DEFAULT_TELEMETRY_DISPLAY: TelemetryDisplayState = { + mode: 'rolling', + windowS: 30, + series: 'chips', +}; + +/** + * Centered time-window mean. Each output sample averages every input sample + * whose timestamp lies within `windowMs / 2` (inclusive) of its own, so the + * smoothed line stays time-aligned with the raw one instead of lagging by half + * a window as a trailing average would. Window edges are inclusive on both + * sides; samples near the start or end of the series average over the shorter + * one-sided neighbourhood that exists. + * + * `samples` must be sorted by `ms` ascending. O(n) via prefix sums. + */ +export function rollingTimeAverage( + samples: readonly TimedSample[], + windowMs: number, +): TimedSample[] { + if (samples.length === 0) return []; + if (!(windowMs > 0)) return samples.map(({ ms, value }) => ({ ms, value })); + const half = windowMs / 2; + const n = samples.length; + const prefix = new Float64Array(n + 1); + for (let i = 0; i < n; i += 1) prefix[i + 1] = prefix[i]! + samples[i]!.value; + + const out: TimedSample[] = Array.from({ length: n }); + let lo = 0; + let hi = 0; + for (let i = 0; i < n; i += 1) { + const center = samples[i]!.ms; + while (samples[lo]!.ms < center - half) lo += 1; + while (hi < n && samples[hi]!.ms <= center + half) hi += 1; + const count = hi - lo; + out[i] = { ms: center, value: (prefix[hi]! - prefix[lo]!) / count }; + } + return out; +} + +/** + * Median gap between consecutive samples, in ms. Used as the alignment + * tolerance when averaging chips that were polled a few ms apart. Falls back + * to `fallbackMs` when fewer than two distinct timestamps exist. + */ +export function estimateSampleIntervalMs( + samples: readonly TimedSample[], + fallbackMs = 1000, +): number { + const gaps: number[] = []; + for (let i = 1; i < samples.length; i += 1) { + const gap = samples[i]!.ms - samples[i - 1]!.ms; + if (gap > 0) gaps.push(gap); + } + if (gaps.length === 0) return fallbackMs; + gaps.sort((a, b) => a - b); + return gaps[Math.floor(gaps.length / 2)]!; +} + +/** + * Mean across several chips at each timestamp of the reference chip (the one + * with the most samples, first on ties). For every reference sample each other + * chip contributes its nearest sample if that sample lies within + * `toleranceMs`; chips with no sample that close are left out of that mean + * rather than interpolated, and `count` records how many contributed. + * + * Every inner array must be sorted by `ms` ascending. + */ +export function meanAcrossSeries( + series: readonly (readonly TimedSample[])[], + toleranceMs: number, +): AggregatedSample[] { + const populated = series.filter((s) => s.length > 0); + if (populated.length === 0) return []; + let reference = populated[0]!; + for (const s of populated) if (s.length > reference.length) reference = s; + const others = populated.filter((s) => s !== reference); + const cursors: number[] = Array.from({ length: others.length }, () => 0); + const tolerance = Math.max(0, toleranceMs); + + return reference.map((ref) => { + let sum = ref.value; + let count = 1; + for (let k = 0; k < others.length; k += 1) { + const other = others[k]!; + let cursor = cursors[k]!; + while ( + cursor + 1 < other.length && + Math.abs(other[cursor + 1]!.ms - ref.ms) <= Math.abs(other[cursor]!.ms - ref.ms) + ) { + cursor += 1; + } + cursors[k] = cursor; + const candidate = other[cursor]!; + if (Math.abs(candidate.ms - ref.ms) <= tolerance) { + sum += candidate.value; + count += 1; + } + } + return { ms: ref.ms, value: sum / count, count }; + }); +} + +/** + * Convert a relative-seconds series (`t` from its own start) to absolute + * milliseconds so it can share an x-axis with wall-clock telemetry. + */ +export function toAbsoluteMs( + points: readonly { t: number; value: number }[], + originMs: number, +): TimedSample[] { + return points.map((p) => ({ ms: originMs + p.t * 1000, value: p.value })); +} + +/** Sample nearest to `ms` (earlier one on a tie), or null when the series is empty. */ +export function nearestSample(samples: readonly TimedSample[], ms: number): TimedSample | null { + if (samples.length === 0) return null; + let lo = 0; + let hi = samples.length - 1; + while (lo < hi) { + const mid = (lo + hi) >> 1; + if (samples[mid]!.ms < ms) lo = mid + 1; + else hi = mid; + } + const after = samples[lo]!; + const before = lo > 0 ? samples[lo - 1]! : after; + return Math.abs(before.ms - ms) <= Math.abs(after.ms - ms) ? before : after; +} diff --git a/packages/app/src/components/gpu-power/types.test.ts b/packages/app/src/components/gpu-power/types.test.ts index cefd8d6ef..1d741a5fa 100644 --- a/packages/app/src/components/gpu-power/types.test.ts +++ b/packages/app/src/components/gpu-power/types.test.ts @@ -3,11 +3,9 @@ import { describe, expect, it } from 'vitest'; import { ALL_METRIC_OPTIONS, computeGpuStats, - detectAnomalies, detectTdpFromArtifactName, getAvailableMetrics, type GpuMetricRow, - GPU_METRIC_OPTIONS, parseCsvData, } from './types'; @@ -47,49 +45,16 @@ describe('parseCsvData', () => { expect(result[1].power).toBe(76.08); }); - it('parses CSV with bare numeric values (no unit suffixes)', () => { - const csv = `${CSV_HEADER} -2024-01-15T10:00:00Z, 0, 250.5, 72, 1980, 1593, 95, 80 -2024-01-15T10:00:00Z, 1, 300.2, 68, 1950, 1593, 88, 75`; - const result = parseCsvData(csv); - expect(result).toHaveLength(2); - expect(result[0].power).toBe(250.5); - expect(result[0].smClock).toBe(1980); - expect(result[1].power).toBe(300.2); - }); - it('returns empty array for header-only CSV', () => { expect(parseCsvData(CSV_HEADER)).toEqual([]); }); - it('returns empty array for empty string', () => { - expect(parseCsvData('')).toEqual([]); - }); - - it('skips rows with insufficient columns', () => { - const csv = `${CSV_HEADER} -2026/03/07 00:20:37.071, 0, 76.78 W -2026/03/07 00:20:37.071, 1, 76.08 W, 29, 345 MHz, 3201 MHz, 0 %, 0 %`; - const result = parseCsvData(csv); - expect(result).toHaveLength(1); - expect(result[0].index).toBe(1); - }); - it('skips rows with NaN values', () => { const csv = `${CSV_HEADER} 2026/03/07 00:20:37.071, abc, 76.78 W, 30, 345 MHz, 3201 MHz, 0 %, 0 %`; expect(parseCsvData(csv)).toEqual([]); }); - it('trims whitespace from values', () => { - const csv = `${CSV_HEADER} - 2026/03/07 00:20:37.071 , 0 , 76.78 W , 30 , 345 MHz , 3201 MHz , 0 % , 0 % `; - const result = parseCsvData(csv); - expect(result).toHaveLength(1); - expect(result[0].timestamp).toBe('2026/03/07 00:20:37.071'); - expect(result[0].power).toBe(76.78); - }); - it('handles Windows-style line endings', () => { const csv = `${CSV_HEADER}\r\n2026/03/07 00:20:37.071, 0, 76.78 W, 30, 345 MHz, 3201 MHz, 0 %, 0 %\r\n`; const result = parseCsvData(csv); @@ -97,19 +62,6 @@ describe('parseCsvData', () => { expect(result[0].power).toBe(76.78); }); - it('parses multiple timestamps for same GPU correctly', () => { - const csv = `${CSV_HEADER} -2026/03/07 00:20:37.071, 0, 76.78 W, 30, 345 MHz, 3201 MHz, 0 %, 0 % -2026/03/07 00:20:38.076, 0, 76.70 W, 30, 345 MHz, 3201 MHz, 0 %, 0 % -2026/03/07 00:20:39.076, 0, 80.49 W, 31, 345 MHz, 3201 MHz, 0 %, 0 %`; - const result = parseCsvData(csv); - expect(result).toHaveLength(3); - expect(result[0].power).toBe(76.78); - expect(result[1].power).toBe(76.7); - expect(result[2].power).toBe(80.49); - expect(result[2].temperature).toBe(31); - }); - it('ignores extra columns beyond the expected 8', () => { const csv = `${CSV_HEADER} 2026/03/07 00:20:37.071, 0, 76.78, 30, 345, 3201, 0, 0, extra1, extra2`; @@ -119,21 +71,6 @@ describe('parseCsvData', () => { expect(result[0].memUtil).toBe(0); }); - it('handles 8-GPU real-world scenario', () => { - const lines = []; - for (let gpu = 0; gpu < 8; gpu++) { - lines.push( - `2026/03/07 00:20:37.071, ${gpu}, ${300 + gpu * 10}, ${65 + gpu}, 1980, 1593, 95, 80`, - ); - } - const csv = `${CSV_HEADER}\n${lines.join('\n')}`; - const result = parseCsvData(csv); - expect(result).toHaveLength(8); - expect(result[7].index).toBe(7); - expect(result[7].power).toBe(370); - expect(result[7].temperature).toBe(72); - }); - // --- AMD amd-smi format --- it('auto-detects and parses AMD amd-smi CSV format', () => { @@ -178,18 +115,6 @@ describe('parseCsvData', () => { expect(result[0].temperature).toBe(38); // falls back to edge }); - it('AMD: handles quoted array fields with embedded commas', () => { - const amdHeader = - 'timestamp,gpu,gfx_activity,umc_activity,mm_activity,vcn_activity,jpeg_activity,socket_power,gfx_0_clk,mem_0_clk,edge,hotspot,mem'; - const csv = `${amdHeader} -1772939616,0,95,80,N/A,"['N/A', 'N/A', 'N/A', 'N/A']","['N/A', 'N/A']",350,1980,901,38,72,65`; - const result = parseCsvData(csv); - expect(result).toHaveLength(1); - expect(result[0].power).toBe(350); - expect(result[0].smClock).toBe(1980); - expect(result[0].temperature).toBe(72); - }); - it('AMD: parses real-world amd-smi row with all columns', () => { // Full amd-smi header with ~100+ columns const amdHeader = @@ -207,26 +132,6 @@ describe('parseCsvData', () => { expect(result[0].temperature).toBe(40); // hotspot }); - it('AMD: parses 8 GPUs from same timestamp', () => { - const amdHeader = - 'timestamp,gpu,gfx_activity,umc_activity,socket_power,gfx_0_clk,mem_0_clk,edge,hotspot,mem'; - const lines = []; - for (let gpu = 0; gpu < 8; gpu++) { - lines.push( - `1772939616,${gpu},${90 + gpu},${70 + gpu},${300 + gpu * 5},${1900 + gpu * 10},901,${35 + gpu},${60 + gpu},${50 + gpu}`, - ); - } - const csv = `${amdHeader}\n${lines.join('\n')}`; - const result = parseCsvData(csv); - expect(result).toHaveLength(8); - expect(result[0].index).toBe(0); - expect(result[7].index).toBe(7); - expect(result[7].power).toBe(335); - expect(result[7].gpuUtil).toBe(97); - expect(result[7].temperature).toBe(67); // hotspot - expect(result[7].smClock).toBe(1970); - }); - it('AMD: skips rows with N/A power', () => { const amdHeader = 'timestamp,gpu,gfx_activity,umc_activity,socket_power,gfx_0_clk,mem_0_clk,edge,hotspot,mem'; @@ -245,7 +150,7 @@ describe('parseCsvData', () => { 1772939616,0,0,0,135,133,901,N/A,N/A,41`; const result = parseCsvData(csv); expect(result).toHaveLength(1); - expect(result[0].temperature).toBe(0); // defaults to 0 when both are N/A + expect(result[0].temperature).toBeUndefined(); }); it('AMD: populates all AMD-specific metric fields', () => { @@ -279,17 +184,6 @@ describe('parseCsvData', () => { expect(result[0].fclk).toBe(1300); expect(result[0].socClk).toBe(28); }); - - it('NVIDIA: does not have AMD-specific fields', () => { - const csv = `${CSV_HEADER} -2026/03/07 00:20:37.071, 0, 300, 65, 1980, 1593, 95, 80`; - const result = parseCsvData(csv); - expect(result).toHaveLength(1); - expect(result[0].edgeTemp).toBeUndefined(); - expect(result[0].memTemp).toBeUndefined(); - expect(result[0].gfxVoltage).toBeUndefined(); - expect(result[0].fclk).toBeUndefined(); - }); }); // --------------------------------------------------------------------------- @@ -297,22 +191,6 @@ describe('parseCsvData', () => { // --------------------------------------------------------------------------- describe('getAvailableMetrics', () => { - it('returns only common metrics for NVIDIA data', () => { - const nvidiaRow: GpuMetricRow = { - timestamp: '2026/03/07 00:20:37.071', - index: 0, - power: 300, - temperature: 65, - smClock: 1980, - memClock: 1593, - gpuUtil: 95, - memUtil: 80, - }; - const metrics = getAvailableMetrics([nvidiaRow]); - const keys = metrics.map((m) => m.key); - expect(keys).toEqual(['power', 'temperature', 'smClock', 'memClock', 'gpuUtil', 'memUtil']); - }); - it('returns common + AMD metrics for AMD data', () => { const amdRow: GpuMetricRow = { timestamp: '2026-03-07T00:00:00Z', @@ -364,11 +242,6 @@ describe('getAvailableMetrics', () => { expect(keys).not.toContain('gfxVoltage'); expect(keys).not.toContain('fclk'); }); - - it('returns common metrics for empty data', () => { - const metrics = getAvailableMetrics([]); - expect(metrics.length).toBe(6); - }); }); // --------------------------------------------------------------------------- @@ -383,54 +256,14 @@ describe('detectTdpFromArtifactName', () => { expect(result).toEqual({ sku: 'H200', tdp: 700 }); }); - it('detects H100 from artifact name', () => { - const result = detectTdpFromArtifactName('gpu_metrics_model_fp8_h100-sxm_0'); - expect(result).toEqual({ sku: 'H100', tdp: 700 }); - }); - it('detects GB200 without matching B200', () => { const result = detectTdpFromArtifactName('gpu_metrics_model_fp4_gb200-nvl72_0'); expect(result).toEqual({ sku: 'GB200', tdp: 1200 }); }); - it('detects GB300 without matching B300', () => { - const result = detectTdpFromArtifactName('gpu_metrics_model_fp4_gb300-nvl72_0'); - expect(result).toEqual({ sku: 'GB300', tdp: 1400 }); - }); - - it('detects B200 when no GB prefix', () => { - const result = detectTdpFromArtifactName('gpu_metrics_model_fp4_b200-sxm_0'); - expect(result).toEqual({ sku: 'B200', tdp: 1000 }); - }); - - it('detects B300 when no GB prefix', () => { - const result = detectTdpFromArtifactName('gpu_metrics_model_fp4_b300-sxm_0'); - expect(result).toEqual({ sku: 'B300', tdp: 1200 }); - }); - - it('detects MI300X from artifact name', () => { - const result = detectTdpFromArtifactName('gpu_metrics_model_fp8_mi300x_0'); - expect(result).toEqual({ sku: 'MI300X', tdp: 750 }); - }); - - it('detects MI325X from artifact name', () => { - const result = detectTdpFromArtifactName('gpu_metrics_model_fp8_mi325x_0'); - expect(result).toEqual({ sku: 'MI325X', tdp: 1000 }); - }); - - it('detects MI355X from artifact name', () => { - const result = detectTdpFromArtifactName('gpu_metrics_model_fp4_mi355x_0'); - expect(result).toEqual({ sku: 'MI355X', tdp: 1400 }); - }); - it('returns null for unrecognized GPU', () => { expect(detectTdpFromArtifactName('gpu_metrics_model_fp8_unknown_0')).toBeNull(); }); - - it('is case-insensitive', () => { - const result = detectTdpFromArtifactName('gpu_metrics_model_fp8_H200-SXM_0'); - expect(result).toEqual({ sku: 'H200', tdp: 700 }); - }); }); // --------------------------------------------------------------------------- @@ -451,187 +284,11 @@ function makeRow( }; } -// --------------------------------------------------------------------------- -// detectAnomalies -// --------------------------------------------------------------------------- - -describe('detectAnomalies', () => { - it('returns empty array for empty data', () => { - expect(detectAnomalies([], 'power')).toEqual([]); - }); - - it('detects statistical outliers via MAD', () => { - const rows: GpuMetricRow[] = []; - for (let i = 0; i < 10; i++) { - rows.push( - makeRow({ - timestamp: `2026/03/07 00:20:${37 + i}.000`, - index: 0, - power: i === 9 ? 600 : 300 + (i % 3), - }), - ); - } - const anomalies = detectAnomalies(rows, 'power'); - expect(anomalies.some((a) => a.type === 'statistical')).toBe(true); - const stat = anomalies.find((a) => a.type === 'statistical')!; - expect(stat.gpuIndex).toBe(0); - expect(stat.value).toBe(600); - }); - - it('detects thermal throttle above 83°C', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 0, temperature: 65 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 0, temperature: 85 }), - makeRow({ timestamp: '2026/03/07 00:20:39.000', index: 0, temperature: 70 }), - ]; - const anomalies = detectAnomalies(rows, 'power'); - expect(anomalies.some((a) => a.type === 'thermal')).toBe(true); - const thermal = anomalies.find((a) => a.type === 'thermal')!; - expect(thermal.value).toBe(85); - }); - - it('detects near-TDP power draw', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 0, power: 300 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 0, power: 650 }), - makeRow({ timestamp: '2026/03/07 00:20:39.000', index: 0, power: 300 }), - ]; - const anomalies = detectAnomalies(rows, 'power', 'gpu_metrics_h200_test'); - expect(anomalies.some((a) => a.type === 'near_tdp')).toBe(true); - }); - - it('does not flag near-TDP without artifact name', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 0, power: 650 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 0, power: 650 }), - makeRow({ timestamp: '2026/03/07 00:20:39.000', index: 0, power: 650 }), - ]; - const anomalies = detectAnomalies(rows, 'power'); - expect(anomalies.some((a) => a.type === 'near_tdp')).toBe(false); - }); - - it('detects utilization drop to 0% after high usage', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 0, gpuUtil: 95 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 0, gpuUtil: 0 }), - makeRow({ timestamp: '2026/03/07 00:20:39.000', index: 0, gpuUtil: 90 }), - ]; - const anomalies = detectAnomalies(rows, 'power'); - expect(anomalies.some((a) => a.type === 'util_drop')).toBe(true); - }); - - it('does not flag all-identical values as anomalies', () => { - const rows: GpuMetricRow[] = []; - for (let i = 0; i < 10; i++) { - rows.push(makeRow({ timestamp: `2026/03/07 00:20:${37 + i}.000`, index: 0, power: 300 })); - } - const anomalies = detectAnomalies(rows, 'power'); - expect(anomalies.filter((a) => a.type === 'statistical')).toHaveLength(0); - }); - - it('deduplicates anomalies at same second', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 0, temperature: 85 }), - makeRow({ timestamp: '2026/03/07 00:20:37.500', index: 0, temperature: 86 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 0, temperature: 65 }), - ]; - const anomalies = detectAnomalies(rows, 'power'); - const thermals = anomalies.filter( - (a) => a.type === 'thermal' && a.gpuIndex === 0 && Math.round(a.seconds) === 0, - ); - expect(thermals).toHaveLength(1); - }); - - it('detects clock drop when SM clock drops >30% below median', () => { - const rows: GpuMetricRow[] = []; - // 9 rows at 1980 MHz, 1 row at 1000 MHz (49% drop) - for (let i = 0; i < 10; i++) { - rows.push( - makeRow({ - timestamp: `2026/03/07 00:20:${37 + i}.000`, - index: 0, - smClock: i === 5 ? 1000 : 1980, - }), - ); - } - const anomalies = detectAnomalies(rows, 'power'); - expect(anomalies.some((a) => a.type === 'clock_drop')).toBe(true); - const drop = anomalies.find((a) => a.type === 'clock_drop')!; - expect(drop.value).toBe(1000); - }); - - it('detects anomalies independently per GPU', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 0, temperature: 85 }), - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 1, temperature: 85 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 0, temperature: 65 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 1, temperature: 65 }), - makeRow({ timestamp: '2026/03/07 00:20:39.000', index: 0, temperature: 65 }), - makeRow({ timestamp: '2026/03/07 00:20:39.000', index: 1, temperature: 65 }), - ]; - const anomalies = detectAnomalies(rows, 'temperature'); - const thermals = anomalies.filter((a) => a.type === 'thermal'); - expect(thermals).toHaveLength(2); - expect(thermals.map((a) => a.gpuIndex).toSorted()).toEqual([0, 1]); - }); - - it('skips GPU groups with fewer than 3 samples for MAD detection', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 0, power: 100 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 0, power: 900 }), - ]; - const anomalies = detectAnomalies(rows, 'power'); - // Only 2 samples — MAD-based statistical detection should be skipped - expect(anomalies.filter((a) => a.type === 'statistical')).toHaveLength(0); - }); - - it('anomaly message includes GPU index and metric value', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 2, temperature: 65 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 2, temperature: 90 }), - makeRow({ timestamp: '2026/03/07 00:20:39.000', index: 2, temperature: 65 }), - ]; - const anomalies = detectAnomalies(rows, 'temperature'); - const thermal = anomalies.find((a) => a.type === 'thermal')!; - expect(thermal.message).toContain('Chip 2'); - expect(thermal.message).toContain('90'); - expect(thermal.message).toContain('83'); - }); - - it('does not flag util_drop when previous utilization is <= 50%', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 0, gpuUtil: 40 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 0, gpuUtil: 0 }), - makeRow({ timestamp: '2026/03/07 00:20:39.000', index: 0, gpuUtil: 50 }), - ]; - const anomalies = detectAnomalies(rows, 'power'); - expect(anomalies.some((a) => a.type === 'util_drop')).toBe(false); - }); -}); - // --------------------------------------------------------------------------- // computeGpuStats // --------------------------------------------------------------------------- describe('computeGpuStats', () => { - it('computes correct statistics for a single GPU', () => { - const rows = [ - makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 0, power: 100 }), - makeRow({ timestamp: '2026/03/07 00:20:38.000', index: 0, power: 200 }), - makeRow({ timestamp: '2026/03/07 00:20:39.000', index: 0, power: 300 }), - makeRow({ timestamp: '2026/03/07 00:20:40.000', index: 0, power: 400 }), - makeRow({ timestamp: '2026/03/07 00:20:41.000', index: 0, power: 500 }), - ]; - const stats = computeGpuStats(rows, 'power'); - expect(stats).toHaveLength(1); - expect(stats[0].gpuIndex).toBe(0); - expect(stats[0].count).toBe(5); - expect(stats[0].min).toBe(100); - expect(stats[0].max).toBe(500); - expect(stats[0].mean).toBe(300); - expect(stats[0].median).toBe(300); - }); - it('returns stats per GPU sorted by index', () => { const rows = [ makeRow({ timestamp: '2026/03/07 00:20:37.000', index: 1, power: 200 }), @@ -695,23 +352,3 @@ describe('computeGpuStats', () => { expect(stats[0].stddev).toBe(0); }); }); - -// --------------------------------------------------------------------------- -// GPU_METRIC_OPTIONS -// --------------------------------------------------------------------------- - -describe('GPU_METRIC_OPTIONS', () => { - it('covers all 6 GpuMetricKey values', () => { - const keys = GPU_METRIC_OPTIONS.map((m) => m.key); - expect(keys).toEqual(['power', 'temperature', 'smClock', 'memClock', 'gpuUtil', 'memUtil']); - }); - - it('each option has non-empty label, unit, and yAxisLabel', () => { - for (const opt of GPU_METRIC_OPTIONS) { - expect(opt.label.length).toBeGreaterThan(0); - expect(opt.unit.length).toBeGreaterThan(0); - expect(opt.yAxisLabel.length).toBeGreaterThan(0); - expect(opt.yAxisLabel).toContain(opt.unit); - } - }); -}); diff --git a/packages/app/src/components/gpu-power/types.ts b/packages/app/src/components/gpu-power/types.ts index 814eb2bb5..6c82d5616 100644 --- a/packages/app/src/components/gpu-power/types.ts +++ b/packages/app/src/components/gpu-power/types.ts @@ -1,24 +1,16 @@ import { HW_REGISTRY } from '@semianalysisai/inferencex-constants'; - -export interface GpuMetricRow { - timestamp: string; - index: number; - power: number; - temperature: number; - smClock: number; - memClock: number; - gpuUtil: number; - memUtil: number; - // AMD-specific optional fields - edgeTemp?: number; - memTemp?: number; - gfxVoltage?: number; - socVoltage?: number; - memVoltage?: number; - fclk?: number; - socClk?: number; - mmActivity?: number; -} +import { + parseAmdTimestamp, + parseNvidiaTimestamp, + splitCsvLine, +} from '@semianalysisai/inferencex-db/etl/gpu-metrics-csv'; +import type { + GpuMetricSampleRow, + GpuMetricSeries, + GpuMetricStatRow, +} from '@semianalysisai/inferencex-db/queries/gpu-metrics'; + +export type GpuMetricRow = GpuMetricSampleRow; export interface GpuPowerRunInfo { id: number; @@ -34,6 +26,8 @@ export interface GpuPowerRunInfo { export interface GpuMetricsArtifact { name: string; data: GpuMetricRow[]; + /** Only database-backed artifacts carry the full-record digest. */ + series?: Omit; } export interface GpuPowerApiResponse { @@ -199,17 +193,24 @@ export const ALL_METRIC_OPTIONS: GpuMetricConfig[] = [...GPU_METRIC_OPTIONS, ... /** * Returns the metric options that have data in the given dataset. - * AMD-specific metrics are only shown when at least one row has a non-undefined value. + * A metric is shown only when at least one row has a finite reading. */ export function getAvailableMetrics(data: GpuMetricRow[]): GpuMetricConfig[] { if (data.length === 0) return GPU_METRIC_OPTIONS; - return ALL_METRIC_OPTIONS.filter((m) => data.some((row) => row[m.key] !== undefined)); + return ALL_METRIC_OPTIONS.filter((m) => data.some((row) => Number.isFinite(row[m.key]))); } /** * Detect GPU SKU from an artifact name and return its TDP in watts. * Artifact names look like: gpu_metrics_dsr1_1k8k_fp8_sglang_tp8_..._h200-nb_0 */ +/** TDP for a known hardware key, e.g. the benchmark point's own `hardware`. */ +export function tdpForHardware(hardware: string | undefined): { sku: string; tdp: number } | null { + const key = hardware?.toLowerCase(); + const entry = key ? HW_REGISTRY[key] : undefined; + return entry ? { sku: key!.toUpperCase(), tdp: entry.tdp } : null; +} + export function detectTdpFromArtifactName( artifactName: string, ): { sku: string; tdp: number } | null { @@ -224,21 +225,6 @@ export function detectTdpFromArtifactName( return null; } -export interface Anomaly { - type: 'statistical' | 'thermal' | 'near_tdp' | 'clock_drop' | 'util_drop'; - label: string; - gpuIndex: number; - seconds: number; - value: number; - message: string; -} - -function median(values: number[]): number { - const sorted = [...values].toSorted((a, b) => a - b); - const mid = Math.floor(sorted.length / 2); - return sorted.length % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid]; -} - function parseTimestampToMs(raw: string): number | null { const d = new Date(raw); if (!isNaN(d.getTime())) return d.getTime(); @@ -247,158 +233,6 @@ function parseTimestampToMs(raw: string): number | null { return null; } -/** - * Detect anomalies in GPU metrics data using MAD-based Modified Z-score - * (per NIST/Iglewicz & Hoaglin, threshold 3.5) plus domain-specific GPU thresholds. - */ -export function detectAnomalies( - data: GpuMetricRow[], - metricKey: GpuMetricKey, - artifactName?: string, -): Anomaly[] { - if (data.length === 0) return []; - - const anomalies: Anomaly[] = []; - const tdpInfo = artifactName ? detectTdpFromArtifactName(artifactName) : null; - - // Find earliest timestamp for seconds calculation - let minTime = Infinity; - for (const row of data) { - const ms = parseTimestampToMs(row.timestamp); - if (ms !== null && ms < minTime) minTime = ms; - } - - // Group by GPU index - const gpuGroups = new Map(); - for (const row of data) { - if (!gpuGroups.has(row.index)) gpuGroups.set(row.index, []); - gpuGroups.get(row.index)!.push(row); - } - - for (const [gpuIndex, rows] of gpuGroups) { - const rawValues = rows.map((r) => r[metricKey]); - // Skip if any values are undefined (metric not available for this vendor) - if (rawValues.some((v) => v === undefined)) continue; - const values = rawValues as number[]; - if (values.length < 3) continue; - - // MAD-based Modified Z-score detection - const med = median(values); - const absDeviations = values.map((v) => Math.abs(v - med)); - const mad = median(absDeviations); - - // Pre-compute SM clock median for clock_drop detection (avoid O(n^2)) - const smMedian = - metricKey === 'smClock' || metricKey === 'power' ? median(rows.map((r) => r.smClock)) : 0; - - for (let i = 0; i < rows.length; i++) { - const row = rows[i]; - const ms = parseTimestampToMs(row.timestamp); - const seconds = ms === null ? i : (ms - minTime) / 1000; - - // Statistical outlier (MAD) - if (mad > 0) { - const modifiedZ = (0.6745 * (values[i] - med)) / mad; - if (Math.abs(modifiedZ) > 3.5) { - anomalies.push({ - type: 'statistical', - label: 'Statistical Outlier', - gpuIndex, - seconds, - value: values[i], - message: `Chip ${gpuIndex} at ${seconds.toFixed(0)}s: ${metricKey} = ${values[i].toFixed(1)} (Modified Z = ${modifiedZ.toFixed(1)}, median = ${med.toFixed(1)})`, - }); - } - } - - // Domain-specific: thermal throttle (temperature > 83°C) - if (row.temperature > 83) { - anomalies.push({ - type: 'thermal', - label: 'Thermal Throttle', - gpuIndex, - seconds, - value: row.temperature, - message: `Chip ${gpuIndex} at ${seconds.toFixed(0)}s: temperature ${row.temperature}°C exceeds throttle threshold (83°C)`, - }); - } - - // Domain-specific: near TDP (power > 90% of TDP) - if (tdpInfo && row.power > tdpInfo.tdp * 0.9) { - anomalies.push({ - type: 'near_tdp', - label: 'Near TDP', - gpuIndex, - seconds, - value: row.power, - message: `Chip ${gpuIndex} at ${seconds.toFixed(0)}s: power ${row.power.toFixed(1)}W is ${((row.power / tdpInfo.tdp) * 100).toFixed(0)}% of TDP (${tdpInfo.tdp}W)`, - }); - } - - // Domain-specific: clock drop (SM clock > 30% below median) - if ( - (metricKey === 'smClock' || metricKey === 'power') && - smMedian > 0 && - row.smClock < smMedian * 0.7 - ) { - anomalies.push({ - type: 'clock_drop', - label: 'Clock Drop', - gpuIndex, - seconds, - value: row.smClock, - message: `Chip ${gpuIndex} at ${seconds.toFixed(0)}s: SM clock ${row.smClock} MHz dropped ${((1 - row.smClock / smMedian) * 100).toFixed(0)}% below median (${smMedian.toFixed(0)} MHz)`, - }); - } - - // Domain-specific: utilization drop (GPU util = 0 after being > 50%) - if (row.gpuUtil === 0 && i > 0 && rows[i - 1].gpuUtil > 50) { - anomalies.push({ - type: 'util_drop', - label: 'Utilization Drop', - gpuIndex, - seconds, - value: 0, - message: `Chip ${gpuIndex} at ${seconds.toFixed(0)}s: utilization dropped to 0% (was ${rows[i - 1].gpuUtil}%)`, - }); - } - } - } - - // Deduplicate: keep only one anomaly per (type, gpuIndex, second) - const seen = new Set(); - const deduped = anomalies.filter((a) => { - const key = `${a.type}_${a.gpuIndex}_${Math.round(a.seconds)}`; - if (seen.has(key)) return false; - seen.add(key); - return true; - }); - - return deduped.toSorted((a, b) => a.seconds - b.seconds); -} - -/** - * Split a CSV line respecting double-quoted fields (which may contain commas). - * Required for AMD amd-smi CSV where array fields like "['N/A', 'N/A']" are quoted. - */ -function splitCsvLine(line: string): string[] { - const result: string[] = []; - let current = ''; - let inQuotes = false; - for (const char of line) { - if (char === '"') { - inQuotes = !inQuotes; - } else if (char === ',' && !inQuotes) { - result.push(current.trim()); - current = ''; - } else { - current += char; - } - } - result.push(current.trim()); - return result; -} - /** * Build a column-index lookup from a CSV header line. * Returns a map of header name → column index. @@ -407,7 +241,7 @@ function buildColumnMap(headerLine: string): Map { const headers = splitCsvLine(headerLine); const map = new Map(); for (let i = 0; i < headers.length; i++) { - map.set(headers[i].toLowerCase(), i); + map.set(headers[i].toLowerCase().replace(/\s*\[.*\]$/u, ''), i); } return map; } @@ -418,40 +252,39 @@ function buildColumnMap(headerLine: string): Map { * clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%] */ function parseNvidiaCsv(lines: string[]): GpuMetricRow[] { + const columns = buildColumnMap(lines[0]); + const col = (name: string) => columns.get(name) ?? -1; + const iTimestamp = col('timestamp'); + const iIndex = col('index'); + const iPower = col('power.draw'); + if (iTimestamp < 0 || iIndex < 0 || iPower < 0) return []; const results: GpuMetricRow[] = []; for (let i = 1; i < lines.length; i++) { const cols = lines[i].split(',').map((c) => c.trim()); - if (cols.length < 8) continue; - - const index = parseInt(cols[1], 10); - const power = parseFloat(cols[2]); - const temperature = parseFloat(cols[3]); - const smClock = parseFloat(cols[4]); - const memClock = parseFloat(cols[5]); - const gpuUtil = parseFloat(cols[6]); - const memUtil = parseFloat(cols[7]); - - if ([index, power, temperature, smClock, memClock, gpuUtil, memUtil].some(isNaN)) continue; + if (cols.length <= Math.max(iTimestamp, iIndex, iPower)) continue; + const index = parseInt(cols[iIndex], 10); + const power = safeFloat(cols[iPower]); + if (!Number.isInteger(index) || index < 0 || power === undefined) continue; results.push({ - timestamp: cols[0], + timestamp: cols[iTimestamp], index, power, - temperature, - smClock, - memClock, - gpuUtil, - memUtil, + temperature: safeFloat(cols[col('temperature.gpu')]), + smClock: safeFloat(cols[col('clocks.current.sm')]), + memClock: safeFloat(cols[col('clocks.current.memory')]), + gpuUtil: safeFloat(cols[col('utilization.gpu')]), + memUtil: safeFloat(cols[col('utilization.memory')]), }); } return results; } -/** Safely parse a float, returning undefined for N/A or unparseable values. */ +/** Missing, unparseable and nonfinite readings stay absent; measured zero is valid. */ function safeFloat(val: string | undefined): number | undefined { if (val === undefined || val === 'N/A' || val === '') return undefined; const n = parseFloat(val); - return isNaN(n) ? undefined : n; + return Number.isFinite(n) ? n : undefined; } /** @@ -484,28 +317,25 @@ function parseAmdCsv(lines: string[], colMap: Map): GpuMetricRow const results: GpuMetricRow[] = []; for (let i = 1; i < lines.length; i++) { const cols = splitCsvLine(lines[i]); - if (cols.length < 3) continue; + if (cols.length <= Math.max(iTimestamp, iGpu, iPower)) continue; const index = parseInt(cols[iGpu], 10); - const power = parseFloat(cols[iPower]); + const power = safeFloat(cols[iPower]); // Prefer hotspot temp, fall back to edge - let temperature = iHotspot >= 0 ? parseFloat(cols[iHotspot]) : NaN; - if (isNaN(temperature) && iEdge >= 0) temperature = parseFloat(cols[iEdge]); - const smClock = iGfxClk >= 0 ? parseFloat(cols[iGfxClk]) : 0; - const memClock = iMemClk >= 0 ? parseFloat(cols[iMemClk]) : 0; - const gpuUtil = iGfxActivity >= 0 ? parseFloat(cols[iGfxActivity]) : 0; - const memUtil = iUmcActivity >= 0 ? parseFloat(cols[iUmcActivity]) : 0; - - if (isNaN(index) || isNaN(power)) continue; - if (isNaN(temperature)) temperature = 0; - - // AMD timestamps are Unix epoch seconds — convert to ms-based string for consistency - const rawTimestamp = cols[iTimestamp]; - const epochSec = parseFloat(rawTimestamp); - const timestamp = - !isNaN(epochSec) && epochSec > 1e9 && epochSec < 1e11 - ? new Date(epochSec * 1000).toISOString() - : rawTimestamp; + const temperature = safeFloat(cols[iHotspot]) ?? safeFloat(cols[iEdge]); + const smClock = safeFloat(cols[iGfxClk]); + const memClock = safeFloat(cols[iMemClk]); + const gpuUtil = safeFloat(cols[iGfxActivity]); + const memUtil = safeFloat(cols[iUmcActivity]); + + if (!Number.isInteger(index) || index < 0 || power === undefined) continue; + + // Round epoch fractions exactly as ingest does before constructing the dedup key. + const timestampMs = parseAmdTimestamp(cols[iTimestamp]); + if (timestampMs === null) continue; + const date = new Date(timestampMs); + if (!Number.isFinite(date.getTime())) continue; + const timestamp = date.toISOString(); // AMD-specific metrics const edgeTemp = iEdge >= 0 ? safeFloat(cols[iEdge]) : undefined; @@ -552,28 +382,27 @@ export function parseCsvData(csvText: string): GpuMetricRow[] { const headerLower = lines[0].toLowerCase(); - // Detect AMD format by checking for amd-smi specific columns - if (headerLower.includes('socket_power') || headerLower.includes('gfx_activity')) { - const colMap = buildColumnMap(lines[0]); - return parseAmdCsv(lines, colMap); - } - - return parseNvidiaCsv(lines); + const rows = + headerLower.includes('socket_power') || headerLower.includes('gfx_activity') + ? parseAmdCsv(lines, buildColumnMap(lines[0])) + : parseNvidiaCsv(lines); + // Normalize one CSV at a time: host-local indices may repeat in other files. + // Keep the first device/timestamp sample, matching the persisted digest. + const seen = new Set(); + return rows.filter((row) => { + // NVIDIA's naive collector clock must not use the reader's local DST rules. + const ms = parseNvidiaTimestamp(row.timestamp) ?? parseTimestampToMs(row.timestamp); + if (ms === null || !Number.isFinite(ms)) return false; + const key = `${row.index}:${ms}`; + if (seen.has(key)) return false; + seen.add(key); + return true; + }); } // --- Statistics --- -export interface GpuStats { - gpuIndex: number; - count: number; - min: number; - max: number; - mean: number; - median: number; - p95: number; - p99: number; - stddev: number; -} +export type GpuStats = Omit; function percentile(sorted: number[], p: number): number { const idx = (p / 100) * (sorted.length - 1); @@ -587,7 +416,7 @@ export function computeGpuStats(data: GpuMetricRow[], metricKey: GpuMetricKey): const groups = new Map(); for (const row of data) { const val = row[metricKey]; - if (val === undefined) continue; + if (val === undefined || !Number.isFinite(val)) continue; if (!groups.has(row.index)) groups.set(row.index, []); groups.get(row.index)!.push(val); } diff --git a/packages/app/src/hooks/api/benchmark-id-query.ts b/packages/app/src/hooks/api/benchmark-id-query.ts index d1540f1d8..0f9adf95b 100644 --- a/packages/app/src/hooks/api/benchmark-id-query.ts +++ b/packages/app/src/hooks/api/benchmark-id-query.ts @@ -49,6 +49,7 @@ export function useByIdQuery( id: number | null, enabled: boolean, select?: (data: T) => TSelected, + freshness?: { staleTime: number; refetchOnWindowFocus: boolean }, ) { const selectNullable = useCallback( (data: T | null): TSelected | null => @@ -67,6 +68,7 @@ export function useByIdQuery( }, enabled, staleTime: STALE_TIME_MS, + ...freshness, // TanStack Query re-runs select when its function identity changes. Keep // this wrapper stable so decoding a large compact payload happens only // when the cached wire data or the caller's selector actually changes. diff --git a/packages/app/src/hooks/api/use-gpu-metrics-point.ts b/packages/app/src/hooks/api/use-gpu-metrics-point.ts new file mode 100644 index 000000000..53ae1dbe8 --- /dev/null +++ b/packages/app/src/hooks/api/use-gpu-metrics-point.ts @@ -0,0 +1,24 @@ +import type { + GpuMetricsPointPayload, + GpuMetricSeries, + GpuMetricStatRow, +} from '@semianalysisai/inferencex-db/queries/gpu-metrics'; + +import { useByIdQuery } from './benchmark-id-query'; + +export type { GpuMetricsPointPayload, GpuMetricSeries, GpuMetricStatRow }; + +/** + * Lazy-fetch the PowerX telemetry linked to one benchmark point. Enabled only + * while the PowerX detail view is open: a series is one 1 s sample per GPU for + * the whole job (hundreds of KB), so it is not paid for on every page load. + */ +export function useGpuMetricsPoint(id: number | null, enabled = false) { + return useByIdQuery( + 'gpu-metrics-point', + id, + enabled && Boolean(id), + undefined, + { staleTime: 0, refetchOnWindowFocus: true }, + ); +} diff --git a/packages/app/src/lib/api-documentation.power.test.ts b/packages/app/src/lib/api-documentation.power.test.ts index 4ecb6ebdb..14986693f 100644 --- a/packages/app/src/lib/api-documentation.power.test.ts +++ b/packages/app/src/lib/api-documentation.power.test.ts @@ -43,10 +43,30 @@ describe('measured-power API documentation', () => { 'max_sample_gap_s', 'producer_sha', 'exporter_image_sha256', + 'cpu', ].toSorted(), ); expect(audit?.required).toBeUndefined(); + // NVL72 CPU-side leg provenance: the enum matches the consumer's sensor kinds. + const cpu = audit?.properties?.cpu; + expect(cpu?.properties?.sensor_kind).toEqual({ + type: 'string', + enum: ['module', 'grace_socket', 'dcgm_cpu_rail'], + }); + expect(cpu?.properties?.reason_codes).toEqual({ type: 'array', items: { type: 'string' } }); + expect(Object.keys(cpu?.properties ?? {}).toSorted()).toEqual( + [ + 'sensor_kind', + 'source', + 'expected_sockets', + 'observed_sockets', + 'sample_row_count', + 'reason_codes', + ].toSorted(), + ); + expect(cpu?.required).toBeUndefined(); + expect(benchmarkRowSchema?.required).not.toContain('power_invalid_reasons'); expect(benchmarkRowSchema?.required).not.toContain('power_audit'); }); @@ -83,6 +103,9 @@ describe('measured-power API documentation', () => { expect(note?.description).toContain('powerValid=strictV2'); expect(note?.description).toContain('power_valid == 1'); expect(note?.description).toContain('power_metric_schema_version == 2'); + expect(note?.description).toContain('cpu_power_valid'); + expect(note?.description).toContain('power_audit.cpu'); + expect(note?.example).toMatchObject({ cpu_power_valid: 1, avg_total_cpu_power_w: 501 }); } const zhNote = getApiDocumentation('zh').schemaNotes.find( (candidate) => candidate.id === 'measured-power', diff --git a/packages/app/src/lib/api-documentation.ts b/packages/app/src/lib/api-documentation.ts index 4974824fb..a6d2f6dab 100644 --- a/packages/app/src/lib/api-documentation.ts +++ b/packages/app/src/lib/api-documentation.ts @@ -240,6 +240,18 @@ const powerMetricDescriptions: Readonly string; strokeWidth?: number; + /** Optional per-series stroke width; falls back to `strokeWidth`, then 2. */ + getStrokeWidth?: (key: string) => number; curve?: d3.CurveFactory; /** Return false to create gaps in the line (e.g., missing data points). */ isDefined?: (d: { x: number; y: number }) => boolean; @@ -61,7 +63,7 @@ export function renderLines( .attr('class', (d) => `line-path line-${d.key}`) .attr('stroke', (d) => config.getColor(d.key)) .attr('stroke-dasharray', (d) => config.getStrokeDasharray?.(d.key) ?? null) - .attr('stroke-width', config.strokeWidth ?? 2) + .attr('stroke-width', (d) => config.getStrokeWidth?.(d.key) ?? config.strokeWidth ?? 2) .attr('d', (d) => lineGenerator(d.points)); } diff --git a/packages/app/src/lib/github-artifacts.test.ts b/packages/app/src/lib/github-artifacts.test.ts index 4a80ae8a4..622f851a6 100644 --- a/packages/app/src/lib/github-artifacts.test.ts +++ b/packages/app/src/lib/github-artifacts.test.ts @@ -123,6 +123,7 @@ describe('extractZipEntries', () => { it('skips non-matching files and continues after parse errors', () => { const zip = new AdmZip(); zip.addFile('good.json', Buffer.from('{"id":1}', 'utf8')); + zip.addFile('upper.JSON', Buffer.from('{"id":2}', 'utf8')); zip.addFile('bad.json', Buffer.from('not json', 'utf8')); zip.addFile('notes.txt', Buffer.from('ignore me', 'utf8')); @@ -136,7 +137,10 @@ describe('extractZipEntries', () => { }, ); - expect(rows).toEqual([{ entryName: 'good.json', payload: { id: 1 } }]); + expect(rows).toEqual([ + { entryName: 'good.json', payload: { id: 1 } }, + { entryName: 'upper.JSON', payload: { id: 2 } }, + ]); expect(parseErrors).toEqual(['bad.json']); }); }); diff --git a/packages/app/src/lib/github-artifacts.ts b/packages/app/src/lib/github-artifacts.ts index 3fe278903..d29162741 100644 --- a/packages/app/src/lib/github-artifacts.ts +++ b/packages/app/src/lib/github-artifacts.ts @@ -120,6 +120,24 @@ export function downloadGithubArtifact(url: string, token: string): Promise boolean, +): Map { + const zip = new AdmZip(buffer); + const files = new Map(); + for (const entry of zip.getEntries()) { + if (entry.isDirectory || !predicate(entry.entryName)) continue; + files.set(entry.entryName, entry.getData().toString('utf8')); + } + return files; +} + export function extractZipEntries( buffer: Buffer, extension: string, @@ -131,7 +149,7 @@ export function extractZipEntries( const rows: T[] = []; for (const entry of zip.getEntries()) { - if (!entry.entryName.endsWith(extension)) { + if (!entry.entryName.toLowerCase().endsWith(extension.toLowerCase())) { continue; } diff --git a/packages/app/src/lib/views-api/docs/extensions.ts b/packages/app/src/lib/views-api/docs/extensions.ts index dc35e0676..150861156 100644 --- a/packages/app/src/lib/views-api/docs/extensions.ts +++ b/packages/app/src/lib/views-api/docs/extensions.ts @@ -22,8 +22,8 @@ const PARAMETER_NOTES: Record = { '快照截止日期,格式为 YYYY-MM-DD。指定 runId 时改为读取该次运行的逻辑快照。', ], runId: [ - 'Positive safe integer workflow run ID. Omit to select the default run; required for live GPU metrics.', - '正安全整数格式的工作流运行 ID。不填时选择默认运行;实时 GPU 指标必须填写。', + 'Positive safe integer workflow run ID. Omit to select the default run; required for GPU metrics.', + '正安全整数格式的工作流运行 ID。不填时选择默认运行;GPU 指标必须填写。', ], precisions: [ 'Comma-separated precision keys; omitted selection uses available curve density. Calculator extensions auto-select the densest official precision and include precisions present in unofficial-run overlays.', @@ -381,9 +381,19 @@ const NEW_VIEWS = { { rows: objects, options: object }, ], 'gpu-metrics': [ - 'Live GPU metrics and statistics', - '实时 GPU 指标与统计', - { runInfo: object, artifacts: strings, rows: objects, stats: objects, rendering: object }, + 'GPU telemetry and full-record statistics', + 'GPU 遥测与全记录统计', + { + runInfo: object, + artifacts: strings, + rows: objects, + stats: { + ...objects, + description: + 'Per-GPU statistics for the selected file/host series, including startup and warmup. Current-version stored digests are authoritative, including an empty or missing metric digest. Outdated or unversioned digests are recomputed read-only from retained DB samples using the current full-record algorithm. Count is the finite-reading count after first-wins timestamp/GPU deduplication. Mean is sample-weighted, P50/P95/P99 use linear interpolation at p*(N-1), and standard deviation divides by N. Values use the selected metric unit; the storage-only metric column is omitted. GPU visibility and chart downsampling do not change this population. These are not serving-window power or J/token.', + }, + rendering: object, + }, ], video: [ 'Published video evidence and tradeoffs', @@ -412,14 +422,14 @@ export const operations: ApiOperation[] = Object.entries(NEW_VIEWS).map( view === 'video' ? ' Only already published artifacts are read. Cell, phase, slot and GPU-basis choices select result evidence and normalized serving rates; x/y/cost/workload filters produce computed tradeoff points. Local bundles and arbitrary URLs are excluded. Responses are no-store.' : view === 'gpu-metrics' - ? ' Live artifact reads are no-store; statistics use all chips and unsampled values, while chart rows respect selected GPU indices.' + ? ' Telemetry reads stored DB series first and falls back to GitHub artifacts when storage is absent. Responses use private, no-store; an upstream database failure remains HTTP 503 rather than an empty dataset. File/host series retain separate identities. Full-record statistics include startup and warmup and use current-version stored per-GPU digests; a current empty or missing metric digest stays empty. Outdated or unversioned digests are recomputed read-only from retained DB samples. Statistics cover every chip in the selected series, while raw rows and charts respect selected GPU indices. Chart downsampling does not alter statistics. Statistics expose the selected metric values without the storage-only metric column. These sample-weighted statistics are separate from serving-window power, J/token and selected-time-window calculations. Line time remains relative to the first sample across all chips; missing metric readings are omitted rather than zero-filled. Correlations require both readings.' : '' }`, `只读${zh},使用仪表板的数据读取和计算函数。未知或重复查询键返回 400;响应包含解析后的参数,保留缺失数据。仅影响样式的控件不作为 API 参数。${ view === 'video' ? ' 仅读取已发布产物。cell、阶段、slot 和 GPU 口径选择对应结果证据,并计算 serving 归一化速率;x/y、成本及工作负载筛选生成权衡图数据点。不读取本地数据包或任意 URL。响应不缓存。' : view === 'gpu-metrics' - ? ' 实时产物读取不缓存;统计量使用所有芯片的未降采样值,图表行则按芯片索引筛选。' + ? ' 遥测优先读取数据库中已存储的序列;缺少存储数据时回退到 GitHub 产物。响应使用 private, no-store;上游数据库故障保留 HTTP 503,不作为空数据返回。各文件、主机的序列身份独立保留。全记录统计包含服务启动与 warmup,已有数据使用当前算法版本的每 GPU 统计摘要;当前摘要为空或缺少所选指标时仍返回空统计数组。旧版本或无版本摘要从保留的 DB 样本只读重算。统计覆盖所选序列的全部芯片,原始数据行和图表则按芯片索引筛选;图表降采样不改变统计。统计项只包含所选指标的数值,不返回数据库内部的 metric 字段。这里按样本计算的统计与 serving-window 功率、J/token 及用户所选时间窗口的计算分别处理。折线时间以所有芯片的首个采样为起点;缺失指标读数会被跳过,不补零。相关性图要求两个指标均有读数。' : '' }`, ), diff --git a/packages/constants/src/metric-keys.test.ts b/packages/constants/src/metric-keys.test.ts index a524a87b4..570da7c5e 100644 --- a/packages/constants/src/metric-keys.test.ts +++ b/packages/constants/src/metric-keys.test.ts @@ -36,9 +36,14 @@ describe('MEASURED_POWER_METRIC_KEYS', () => { 'peak_temp_c', 'avg_util_pct', 'avg_mem_used_mb', + 'avg_cpu_socket_power_w', + 'avg_total_cpu_power_w', + 'total_cpu_energy_j', + 'avg_total_module_power_w', + 'total_module_energy_j', ]), ); - expect(MEASURED_POWER_METRIC_KEYS.size).toBe(19); + expect(MEASURED_POWER_METRIC_KEYS.size).toBe(24); }); it('never contains the contract discriminators or invalid-verdict companion fields', () => { @@ -46,6 +51,7 @@ describe('MEASURED_POWER_METRIC_KEYS', () => { for (const key of [ 'power_valid', 'power_metric_schema_version', + 'cpu_power_valid', 'power_invalid_reasons', 'power_audit', ]) { @@ -69,8 +75,13 @@ describe('POWER_METRIC_KEYS', () => { // The public API documentation types every one of these keys on // BenchmarkRow.metrics, so membership changes are contract changes. expect(new Set(POWER_METRIC_KEYS)).toEqual( - new Set(['power_valid', 'power_metric_schema_version', ...MEASURED_POWER_METRIC_KEY_LIST]), + new Set([ + 'power_valid', + 'power_metric_schema_version', + 'cpu_power_valid', + ...MEASURED_POWER_METRIC_KEY_LIST, + ]), ); - expect(POWER_METRIC_KEYS).toHaveLength(21); + expect(POWER_METRIC_KEYS).toHaveLength(27); }); }); diff --git a/packages/constants/src/metric-keys.ts b/packages/constants/src/metric-keys.ts index 3c983ccb3..922596f97 100644 --- a/packages/constants/src/metric-keys.ts +++ b/packages/constants/src/metric-keys.ts @@ -1,8 +1,36 @@ /** - * Power, energy, and GPU telemetry withheld at ingest and display when the - * normalized `power_valid` verdict is 0. Contract and diagnostic fields are - * excluded so the invalid verdict remains auditable. Add new measured fields - * here; `METRIC_KEYS` derives from this list. + * NVL72 Grace-side and compute-module measurements from the srt-slurm CPU power + * leg (ACPI hwmon), integrated over the same formal window as GPU energy. They + * are withheld on their own `cpu_power_valid` verdict, never on `power_valid`: + * the leg has its own sensors, window bracketing and gap checks, so a failed + * GPU leg says nothing about them and vice versa. Unlike `power_valid`, no + * legacy rows predate the verdict, so an absent verdict withholds them too. + * avg_cpu_socket_power_w: mean over sockets of each socket's window-mean Grace-side W + * avg_total_cpu_power_w: sum over sockets of window-mean Grace-side W + * total_cpu_energy_j: Grace-side energy over the window, all sockets + * avg_total_module_power_w / total_module_energy_j: whole compute module + * (Grace + GPUs + HBM + LPDDR5X + regulator loss), only when + * the module sensor exists on every socket + */ +export const CPU_SIDE_POWER_METRIC_KEY_LIST = [ + 'avg_cpu_socket_power_w', + 'avg_total_cpu_power_w', + 'total_cpu_energy_j', + 'avg_total_module_power_w', + 'total_module_energy_j', +] as const; + +export const CPU_SIDE_POWER_METRIC_KEYS: ReadonlySet = new Set( + CPU_SIDE_POWER_METRIC_KEY_LIST, +); + +/** + * Every measured power, energy, and telemetry field. GPU-side fields are + * withheld at ingest and display when the normalized `power_valid` verdict is + * 0; the `CPU_SIDE_POWER_METRIC_KEY_LIST` subset follows `cpu_power_valid` + * instead. Contract and diagnostic fields are excluded so the invalid verdict + * remains auditable. Add new measured fields here; `METRIC_KEYS` derives from + * this list. */ export const MEASURED_POWER_METRIC_KEY_LIST = [ // measured power / energy (emitted by runner's aggregate_power.py) @@ -46,6 +74,7 @@ export const MEASURED_POWER_METRIC_KEY_LIST = [ 'peak_temp_c', 'avg_util_pct', 'avg_mem_used_mb', + ...CPU_SIDE_POWER_METRIC_KEY_LIST, ] as const; export const MEASURED_POWER_METRIC_KEYS: ReadonlySet = new Set( @@ -65,7 +94,11 @@ export const POWER_METRIC_KEYS = [ // joules_per_* field as whole-deployment energy 'power_valid', 'power_metric_schema_version', + // cpu_power_valid: numeric 1/0 verdict for the NVL72 CPU-side leg, independent + // of power_valid; anything but 1 withholds the CPU-side keys + 'cpu_power_valid', // measured power / energy / telemetry values, withheld when power_valid = 0 + // (GPU side) or cpu_power_valid != 1 (CPU side) ...MEASURED_POWER_METRIC_KEY_LIST, ] as const; diff --git a/packages/constants/src/tables.ts b/packages/constants/src/tables.ts index 684f310bc..d1bcd6a0b 100644 --- a/packages/constants/src/tables.ts +++ b/packages/constants/src/tables.ts @@ -5,6 +5,10 @@ export const TABLE_NAMES = { agenticTraceReplay: 'agentic_trace_replay', benchmarkResults: 'benchmark_results', serverLogs: 'server_logs', + gpuMetricSeries: 'gpu_metric_series', + gpuMetricSamples: 'gpu_metric_samples', + gpuMetricGpuStats: 'gpu_metric_gpu_stats', + benchmarkResultGpuMetrics: 'benchmark_result_gpu_metrics', runStats: 'run_stats', evalResults: 'eval_results', evalSamples: 'eval_samples', diff --git a/packages/db/migrations/016_gpu_metrics.sql b/packages/db/migrations/016_gpu_metrics.sql new file mode 100644 index 000000000..8c3f52ac5 --- /dev/null +++ b/packages/db/migrations/016_gpu_metrics.sql @@ -0,0 +1,99 @@ +-- PowerX telemetry digest. +-- +-- The producer samples nvidia-smi / amd-smi once per second for the lifetime +-- of every benchmark job and uploads the CSV as a `gpu_metrics_` +-- artifact next to `bmk_`. Until now the app downloaded and parsed +-- those artifacts from GitHub on every page view, and lost them entirely once +-- GitHub's 90-day artifact retention expired. These tables move that work to +-- ingest time: raw samples are kept at full resolution, per-GPU summary +-- statistics are digested once, and each benchmark point is linked to the +-- series that was recorded while it ran. + +create table gpu_metric_series ( + id bigserial primary key, + workflow_run_id bigint not null references workflow_runs(id) on delete cascade, + -- Full GitHub artifact name, including the runner-pool/attempt suffix. + artifact_name text not null, + -- Artifact name with the gpu_metrics_ prefix removed; pairs with bmk_. + config_key text not null, + -- CSV path relative to the extracted artifact root (multinode uploads can + -- carry one CSV per serving node). + file_name text not null, + vendor text not null, + csv_sha256 text not null, + sample_interval_s real, + sample_count integer not null, + gpu_count smallint not null, + started_at timestamptz not null, + ended_at timestamptz not null, + -- Parsed sidecars: gpu_metrics_context.json, gpu_metrics_identity.*, + -- gpu_metrics_energy_{start,end}.csv. Kept verbatim for provenance. + sidecars jsonb not null default '{}'::jsonb, + ingested_at timestamptz not null default now(), + + constraint gpu_metric_series_vendor_known check (vendor in ('nvidia', 'amd')), + constraint gpu_metric_series_artifact_nonempty check (artifact_name <> ''), + constraint gpu_metric_series_file_nonempty check (file_name <> ''), + constraint gpu_metric_series_sample_count_non_neg check (sample_count >= 0), + constraint gpu_metric_series_gpu_count_non_neg check (gpu_count >= 0), + constraint gpu_metric_series_window_ordered check (ended_at >= started_at), + constraint gpu_metric_series_unique unique (workflow_run_id, artifact_name, file_name) +); + +create index gpu_metric_series_run_idx on gpu_metric_series (workflow_run_id); + +-- One row per (GPU, sample). Vendor-specific columns stay null when the +-- collector does not report them. `real` keeps the row narrow; the source +-- telemetry has at most three significant decimals. +create table gpu_metric_samples ( + series_id bigint not null references gpu_metric_series(id) on delete cascade, + gpu_index smallint not null, + sampled_at timestamptz not null, + power_w real, + temperature_c real, + sm_clock_mhz real, + mem_clock_mhz real, + gpu_util_pct real, + mem_util_pct real, + edge_temp_c real, + mem_temp_c real, + gfx_voltage_mv real, + soc_voltage_mv real, + mem_voltage_mv real, + fclk_mhz real, + socclk_mhz real, + mm_activity_pct real, + + primary key (series_id, gpu_index, sampled_at) +); + +-- Ingest-time digest so readers never rescan samples for summary cards. +create table gpu_metric_gpu_stats ( + series_id bigint not null references gpu_metric_series(id) on delete cascade, + gpu_index smallint not null, + metric text not null, + sample_count integer not null, + min_value real not null, + max_value real not null, + mean_value real not null, + median_value real not null, + p95_value real not null, + p99_value real not null, + stddev_value real not null, + + constraint gpu_metric_gpu_stats_metric_nonempty check (metric <> ''), + constraint gpu_metric_gpu_stats_sample_count_positive check (sample_count > 0), + primary key (series_id, gpu_index, metric) +); + +-- Benchmark point ↔ telemetry series. A point can reference several series +-- when a multinode artifact ships one CSV per node. +create table benchmark_result_gpu_metrics ( + benchmark_result_id bigint not null references benchmark_results(id) on delete cascade, + series_id bigint not null references gpu_metric_series(id) on delete cascade, + + primary key (benchmark_result_id, series_id) +); + +create index benchmark_result_gpu_metrics_series_idx + on benchmark_result_gpu_metrics (series_id); diff --git a/packages/db/migrations/017_gpu_metric_stats_version.sql b/packages/db/migrations/017_gpu_metric_stats_version.sql new file mode 100644 index 000000000..68f7104b3 --- /dev/null +++ b/packages/db/migrations/017_gpu_metric_stats_version.sql @@ -0,0 +1,3 @@ +-- Zero means unversioned, not computed by the current algorithm. +ALTER TABLE gpu_metric_series + ADD COLUMN IF NOT EXISTS stats_version integer NOT NULL DEFAULT 0; diff --git a/packages/db/package.json b/packages/db/package.json index ba2589286..4ef2fe986 100644 --- a/packages/db/package.json +++ b/packages/db/package.json @@ -27,6 +27,7 @@ "db:backfill-atom-kv-capacity": "bun --env-file=../../.env src/backfill-atom-kv-capacity.ts", "db:backfill-chart-series": "bun --env-file=../../.env src/backfill-chart-series.ts", "db:backfill-dataset-stats": "bun --env-file=../../.env src/backfill-dataset-stats.ts", + "db:backfill-gpu-metrics": "bun --env-file=../../.env src/backfill-gpu-metrics.ts", "db:backfill-full-response-interactivity": "bun --env-file=../../.env src/backfill-full-response-interactivity.ts", "db:backfill-request-timeline": "bun --env-file=../../.env src/backfill-request-timeline.ts", "db:backfill-runtime-metadata": "bun --env-file=../../.env src/backfill-runtime-metadata.ts", diff --git a/packages/db/src/apply-overrides.ts b/packages/db/src/apply-overrides.ts index 6ac41680b..62188b705 100644 --- a/packages/db/src/apply-overrides.ts +++ b/packages/db/src/apply-overrides.ts @@ -26,6 +26,13 @@ import { type Sql, createAdminSql, refreshLatestBenchmarks } from './etl/db-util import { jsonbParam } from './lib/backfill-runner.js'; import { planBenchmarkPointBackfill } from './lib/benchmark-point-backfill.js'; import { selectRunOverrides } from './lib/run-override-selection.js'; +import { + type TelemetryPurgeCounts, + countRunTelemetry, + deleteRunTelemetry, + describeTelemetry, + unlinkPointTelemetry, +} from './lib/telemetry-purge.js'; import { type BenchmarkPointBackfill, type ChangelogBackfill, @@ -315,6 +322,7 @@ interface PurgeTarget { stats: number; evals: number; changelogs: number; + telemetry: TelemetryPurgeCounts; } /** @@ -358,17 +366,21 @@ async function previewPurge( ); } - const [[bmk], [stats], [evals], [changelogs], [logs]] = await Promise.all([ + const [[bmk], [stats], [evals], [changelogs], [logs], telemetry] = await Promise.all([ sql`SELECT count(*)::int AS n FROM benchmark_results WHERE workflow_run_id = ANY(${wrIds})`, sql`SELECT count(*)::int AS n FROM run_stats WHERE workflow_run_id = ANY(${wrIds})`, sql`SELECT count(*)::int AS n FROM eval_results WHERE workflow_run_id = ANY(${wrIds})`, sql`SELECT count(*)::int AS n FROM changelog_entries WHERE workflow_run_id = ANY(${wrIds})`, sql`SELECT count(DISTINCT server_log_id)::int AS n FROM benchmark_results WHERE workflow_run_id = ANY(${wrIds}) AND server_log_id IS NOT NULL`, + countRunTelemetry(sql, wrIds), ]); console.log( ` ${bmk.n} benchmarks, ${logs.n} server_logs, ${stats.n} run_stats, ${evals.n} evals, ${changelogs.n} changelogs`, ); + // Surfaced separately: past GitHub's 90-day artifact retention these samples + // are the only copy, so the operator should see them before confirming. + if (telemetry.series > 0) console.log(` ${describeTelemetry(telemetry)}`); return { githubRunId, @@ -378,6 +390,7 @@ async function previewPurge( stats: stats.n, evals: evals.n, changelogs: changelogs.n, + telemetry, }; } @@ -404,6 +417,14 @@ async function purgeBenchmarkResults(tx: Sql, resultIds: number[]): Promise 0) + console.log(` unlinked ${unlinked} gpu_metric_series link(s); series kept with the run.`); + await tx`DELETE FROM benchmark_results WHERE id = ANY(${resultIds})`; const sIds = logRows.map((r) => r.id as number); @@ -493,6 +514,11 @@ async function purge(wrIds: number[]): Promise { await tx`DELETE FROM eval_results WHERE workflow_run_id = ANY(${wrIds})`; await tx`DELETE FROM changelog_entries WHERE workflow_run_id = ANY(${wrIds})`; + // Telemetry too. The workflow_runs delete below would cascade it away anyway, + // but silently; deleting it here reports the cost of an irreversible loss. + const telemetry = await deleteRunTelemetry(tx, wrIds); + if (telemetry.series > 0) console.log(` deleted ${describeTelemetry(telemetry)}.`); + // Parent last (target the specific workflow_runs rows so partial purges // leave sibling attempts of the same github_run_id intact) await tx`DELETE FROM workflow_runs WHERE id = ANY(${wrIds})`; diff --git a/packages/db/src/backfill-gpu-metrics.ts b/packages/db/src/backfill-gpu-metrics.ts new file mode 100644 index 000000000..cb898f0d8 --- /dev/null +++ b/packages/db/src/backfill-gpu-metrics.ts @@ -0,0 +1,675 @@ +/** + * Backfill PowerX telemetry (`gpu_metrics_` artifacts) into the + * migration-016 tables for runs that were ingested before the CI path + * digested them. + * + * GitHub keeps run artifacts for 90 days and the GCS mirror only covers + * scheduled/push runs on main, so the reachable history is bounded by GitHub + * retention; runs GitHub has since deleted are reported and skipped. Each + * gpu_metrics artifact is paired with its exact `bmk_` + * (or `bmk_agentic_`) sibling, the raw rows are mapped through the + * production mapper, and the series is linked to those persisted points. + * + * Usage: + * bun run --cwd packages/db db:backfill-gpu-metrics --run 34557177019 --yes + * bun run --cwd packages/db db:backfill-gpu-metrics --all --yes + * bun run --cwd packages/db db:backfill-gpu-metrics --all --since 2026-08-01 --dry-run + * bun run --cwd packages/db db:backfill-gpu-metrics --all --force --limit 20 --yes + * + * bun run --cwd packages/db db:backfill-gpu-metrics --stats-only --all --dry-run + * bun run --cwd packages/db db:backfill-gpu-metrics --stats-only --run 34557177019 --yes + * + * --stats-only upgrades outdated digests from DB samples without GitHub access. + * It includes all retained attempts and dates unless explicitly filtered. + * + * Runs that already have at least one stored series are skipped unless + * --force is passed. --parallel N (default 4) bounds concurrent artifact + * downloads within one run. + */ + +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { hasNoSslFlag } from './cli-utils.js'; +import { AsyncSemaphore } from './etl/async-semaphore.js'; +import { createAdminSql } from './etl/db-utils.js'; +import { ingestGpuMetricsArtifact, refreshGpuMetricStats } from './etl/gpu-metrics-ingest.js'; +import { readPowerAuditValidations } from './etl/gpu-metrics-artifacts.js'; +import { recoveredPowerAudit } from './etl/power-audit-validations.js'; +import { + benchmarkPublicationIdentity, + stablePowerPointIdentity, + type PowerPublicationManifest, +} from './etl/power-publication.js'; +import { + readTelemetryReceipt, + summarizeTelemetryReceipt, + telemetryArtifactsForAttempt, + type TelemetryObservation, + type TelemetryReceipt, +} from './etl/telemetry-receipt.js'; +import { retryArtifactOperation } from './lib/artifact-retry.js'; +import { + checkpointBenchmarkRefresh, + refreshBackfillBenchmarks, + type BenchmarkAuditUpdate, +} from './lib/backfill-benchmark-refresh.js'; +import { + confirmProceed, + listBackfillRunArtifacts, + parseLimitForceFlags, + runBackfillMain, +} from './lib/backfill-runner.js'; +import { + filterPurgedBenchmarkRows, + findBenchmarkResultIds, + readMappedBenchmarkRows, +} from './lib/benchmark-result-lookup.js'; +import { downloadArtifact, fetchRunMeta } from './lib/github-artifacts.js'; +import { + pairGpuMetricsArtifacts, + findOutdatedGpuMetricSeries, + collectMissingTelemetryExpectations, + type GpuMetricsArtifactPair, +} from './lib/gpu-metrics-backfill.js'; +import { repositoryFromRunUrl } from './lib/runtime-metadata-artifacts.js'; + +const DEFAULT_REPO = 'SemiAnalysisAI/InferenceX'; +const GITHUB_RETENTION_DAYS = 90; +const sql = createAdminSql({ noSsl: hasNoSslFlag(), max: 4, onnotice: () => {} }); + +interface CandidateRun { + id: number; + github_run_id: number; + run_attempt: number; + html_url: string | null; + date: string; + series_count: number; +} + +interface BackfillFlags { + all: boolean; + dryRun: boolean; + run: number | null; + fromRun: number | null; + since: string | null; + parallel: number; + attempt: number | null; + artifact: string | null; + receipt: string | null; + refreshCacheOnly: boolean; + statsOnly: boolean; +} + +function positiveIntFlag(flag: string): number | null { + const index = process.argv.indexOf(flag); + if (index === -1) return null; + const raw = process.argv[index + 1]; + if (!raw || !/^\d+$/u.test(raw) || Number(raw) <= 0) { + throw new Error(`${flag} requires a positive integer`); + } + return Number(raw); +} + +function stringFlag(flag: string): string | null { + const index = process.argv.indexOf(flag); + if (index === -1) return null; + const value = process.argv[index + 1]; + if (!value || value.startsWith('--')) throw new Error(`${flag} requires a value`); + return value; +} + +function parseFlags(): BackfillFlags { + const sinceIndex = process.argv.indexOf('--since'); + const since = sinceIndex === -1 ? null : (process.argv[sinceIndex + 1] ?? null); + if (sinceIndex !== -1 && (!since || !/^\d{4}-\d{2}-\d{2}$/u.test(since))) { + throw new Error('--since requires a YYYY-MM-DD date'); + } + return { + all: process.argv.includes('--all'), + dryRun: process.argv.includes('--dry-run'), + run: positiveIntFlag('--run'), + fromRun: positiveIntFlag('--from-run'), + since, + parallel: positiveIntFlag('--parallel') ?? 4, + attempt: positiveIntFlag('--attempt'), + artifact: stringFlag('--artifact'), + receipt: stringFlag('--receipt'), + refreshCacheOnly: process.argv.includes('--refresh-cache-only'), + statsOnly: process.argv.includes('--stats-only'), + }; +} + +function isWithinGithubRetention(date: string): boolean { + const ageMs = Date.now() - new Date(date).getTime(); + return ageMs <= GITHUB_RETENTION_DAYS * 24 * 60 * 60 * 1000; +} + +async function loadCandidateRuns( + flags: BackfillFlags, + limit: number | null, + force: boolean, +): Promise { + const cutoff = new Date(Date.now() - GITHUB_RETENTION_DAYS * 24 * 60 * 60 * 1000) + .toISOString() + .slice(0, 10); + const since = flags.since ?? cutoff; + const rows = await sql` + select wr.id, wr.github_run_id, wr.run_attempt, wr.html_url, wr.date::text as date, + (select count(*)::int from gpu_metric_series s where s.workflow_run_id = wr.id) as series_count + from workflow_runs wr + where exists (select 1 from benchmark_results br where br.workflow_run_id = wr.id) + and (${flags.run}::bigint is null or wr.github_run_id = ${flags.run}) + and (${flags.fromRun}::bigint is null or wr.github_run_id >= ${flags.fromRun}) + and (${flags.run}::bigint is not null or wr.date >= ${since}::date) + and ((${flags.attempt}::integer is not null and wr.run_attempt = ${flags.attempt}) or + (${flags.attempt}::integer is null and not exists ( + select 1 from workflow_runs newer where newer.github_run_id = wr.github_run_id + and newer.run_attempt > wr.run_attempt))) + order by wr.date desc, wr.github_run_id desc + `; + const candidates = rows + .map((row) => ({ + ...row, + id: Number(row.id), + github_run_id: Number(row.github_run_id), + series_count: Number(row.series_count), + })) + .filter((row) => force || flags.run !== null || row.series_count === 0); + return limit === null ? candidates : candidates.slice(0, limit); +} + +type PairOutcome = + | { + kind: 'ingested'; + seriesCount: number; + samplesInserted: number; + pointsLinked: number; + expectationsUnknown: boolean; + metadataUpdatedBenchmarkResultIds: number[]; + } + | { kind: 'unmatched' } + | { kind: 'failed' }; + +/** Download one gpu_metrics/bmk pair, resolve its points, and persist the series. */ +async function processPair( + run: CandidateRun, + pair: GpuMetricsArtifactPair, + tempDir: string, + observations: Map, + expectationErrors: NonNullable, + uniqueFallbacks: Map, + checkpointMetadata: (updates: BenchmarkAuditUpdate[]) => Promise, +): Promise { + let benchmarkDir: string | null = null; + let gpuMetricsDir: string | null = null; + const pointKeys: string[] = []; + let expectationsUnknown = false; + try { + benchmarkDir = await retryArtifactOperation(`downloading ${pair.benchmarks.name}`, () => + downloadArtifact(pair.benchmarks, tempDir), + ); + const mappedRows = await filterPurgedBenchmarkRows( + sql, + run, + readMappedBenchmarkRows(benchmarkDir, (error) => { + expectationsUnknown = true; + expectationErrors.push({ + benchmarkArtifact: pair.benchmarks.name, + artifactNames: [pair.gpuMetrics.name], + error, + }); + }), + ); + for (const row of mappedRows) { + const identity = benchmarkPublicationIdentity(row); + const key = stablePowerPointIdentity(identity); + pointKeys.push(key); + observations.set(key, { identity, artifactNames: [pair.gpuMetrics.name], produced: true }); + } + const matchedIds: number[] = []; + const mappedPoints: { ids: number[]; identity: Record }[] = []; + for (const row of mappedRows) { + const ids = await findBenchmarkResultIds(sql, run, [row], (id) => + uniqueFallbacks.set(stablePowerPointIdentity(benchmarkPublicationIdentity(row)), id), + ); + if (ids.length === 0) throw new Error(`${pair.gpuMetrics.name}: no matching benchmark rows`); + matchedIds.push(...ids); + if (row.benchmarkType === 'agentic_traces') + mappedPoints.push({ ids, identity: benchmarkPublicationIdentity(row) }); + } + const resultIds = [...new Set(matchedIds)]; + if (resultIds.length === 0) { + if (expectationsUnknown) return { kind: 'failed' }; + console.warn(` [WARN] ${pair.gpuMetrics.name}: no matching benchmark rows`); + return { kind: 'unmatched' }; + } + gpuMetricsDir = await retryArtifactOperation(`downloading ${pair.gpuMetrics.name}`, () => + downloadArtifact(pair.gpuMetrics, tempDir), + ); + const validations = Object.entries( + readPowerAuditValidations(gpuMetricsDir, pair.gpuMetrics.name), + ) + .map(([source, validation]) => ({ + powerAudit: recoveredPowerAudit(source, validation), + conc: (validation.selected_window as Record | undefined)?.concurrency, + })) + .filter((validation) => validation.powerAudit !== null); + const auditUpdates: BenchmarkAuditUpdate[] = []; + for (const point of mappedPoints) { + const candidates = validations.filter( + (validation) => validation.conc === point.identity.conc, + ); + if (candidates.length !== 1) continue; + const powerAudit = candidates[0]!.powerAudit; + if (powerAudit) + auditUpdates.push( + ...point.ids.map((benchmarkResultId) => ({ + benchmarkResultId, + identity: point.identity, + powerAudit, + })), + ); + } + await checkpointMetadata(auditUpdates); + const ingested = await ingestGpuMetricsArtifact(sql, { + workflowRunId: run.id, + artifact: { artifactName: pair.gpuMetrics.name, artifactDir: gpuMetricsDir }, + benchmarkResultIds: resultIds, + }); + if (ingested.seriesIds.length === 0) { + throw new Error(`${pair.gpuMetrics.name}: no parseable gpu_metrics CSV`); + } + return { + kind: 'ingested', + seriesCount: ingested.seriesIds.length, + samplesInserted: ingested.samplesInserted, + pointsLinked: resultIds.length, + expectationsUnknown, + metadataUpdatedBenchmarkResultIds: ingested.metadataUpdatedBenchmarkResultIds, + }; + } catch (error) { + if (pointKeys.length === 0) + expectationErrors.push({ + benchmarkArtifact: pair.benchmarks.name, + artifactNames: [pair.gpuMetrics.name], + error: error instanceof Error ? error.message : String(error), + }); + for (const key of pointKeys) { + const observation = observations.get(key)!; + observation.error = error instanceof Error ? error.message : String(error); + } + console.error(` ✗ run ${run.github_run_id} artifact ${pair.gpuMetrics.name}:`, error); + return { kind: 'failed' }; + } finally { + if (benchmarkDir) fs.rmSync(benchmarkDir, { recursive: true, force: true }); + if (gpuMetricsDir) fs.rmSync(gpuMetricsDir, { recursive: true, force: true }); + } +} + +async function main(): Promise { + const flags = parseFlags(); + const { limit, force } = parseLimitForceFlags(); + if (!flags.all && flags.run === null) { + throw new Error('Pass --run or --all'); + } + if ((flags.attempt !== null || flags.artifact || flags.receipt) && flags.run === null) + throw new Error('--attempt, --artifact and --receipt require --run'); + if ( + flags.refreshCacheOnly && + (!flags.run || !flags.receipt || flags.artifact || flags.all || flags.dryRun) + ) + throw new Error( + '--refresh-cache-only requires --run and --receipt, without --all, --artifact or --dry-run', + ); + if (flags.refreshCacheOnly && !fs.existsSync(flags.receipt!)) + throw new Error('--refresh-cache-only requires an existing receipt'); + + if (flags.statsOnly) { + if (flags.refreshCacheOnly || flags.receipt || force) + throw new Error( + '--stats-only cannot be combined with --refresh-cache-only, --receipt or --force', + ); + const series = await findOutdatedGpuMetricSeries(sql, flags, limit); + console.log(`${series.length} outdated digest(s); --limit counts series in --stats-only mode`); + if (flags.dryRun) { + console.table(series); + return; + } + if ( + series.length === 0 || + !(await confirmProceed('Recompute only these stored telemetry digests?')) + ) + return; + for (const row of series) { + try { + const updated = await refreshGpuMetricStats(sql, Number(row.id)); + console.log( + `series ${row.id} run ${row.github_run_id} attempt ${row.run_attempt}: ${updated ? 'updated' : 'current'}`, + ); + } catch (error) { + process.exitCode = 1; + console.error( + `series ${row.id} failed; retry --stats-only --run ${row.github_run_id} --attempt ${row.run_attempt} --artifact ${row.artifact_name} --yes`, + error, + ); + } + } + return; + } + + console.log('=== backfill-gpu-metrics ==='); + const runs = await loadCandidateRuns(flags, limit, force); + const staleRuns = runs.filter((run) => !isWithinGithubRetention(run.date)).length; + const staleNote = + staleRuns > 0 + ? ` (${staleRuns} older than GitHub's ${GITHUB_RETENTION_DAYS}-day retention)` + : ''; + console.log(` ${runs.length} candidate run(s)${staleNote}`); + if (runs.length === 0) { + if (flags.attempt !== null || flags.artifact || flags.receipt) + throw new Error( + `No persisted benchmark target for run ${flags.run} attempt ${flags.attempt ?? 'latest'}`, + ); + console.log('\n Nothing to do.'); + return; + } + + if (flags.dryRun) { + let pairedRuns = 0; + let pairs = 0; + let goneRuns = 0; + let failedRuns = 0; + for (const run of runs) { + try { + const repository = repositoryFromRunUrl(run.html_url) ?? DEFAULT_REPO; + const artifacts = await listBackfillRunArtifacts(repository, run.github_run_id); + if (artifacts === null) { + goneRuns++; + console.log(` run ${run.github_run_id} (${run.date}): gone from GitHub`); + continue; + } + const runPairs = pairGpuMetricsArtifacts( + telemetryArtifactsForAttempt( + artifacts, + fetchRunMeta(repository, String(run.github_run_id)), + run.run_attempt, + ), + ).filter((pair) => !flags.artifact || pair.gpuMetrics.name === flags.artifact); + if (runPairs.length > 0) pairedRuns++; + pairs += runPairs.length; + console.log(` run ${run.github_run_id} (${run.date}): ${runPairs.length} pair(s)`); + } catch (error) { + if (flags.run !== null) throw error; + failedRuns++; + console.error( + ` run ${run.github_run_id} (${run.date}): ${error instanceof Error ? error.message : String(error)}`, + ); + } + } + console.log( + `\n=== dry run: ${runs.length} run(s), ${pairedRuns} with pairs, ${pairs} pair(s), ` + + `${goneRuns} gone from GitHub, ${failedRuns} failed ===`, + ); + if (failedRuns > 0) process.exitCode = 1; + return; + } + + if ( + !(await confirmProceed( + flags.refreshCacheOnly + ? 'Only the recorded benchmark metadata cache will be refreshed.' + : `${runs.length} workflow run(s) will be checked for gpu_metrics.`, + )) + ) { + return; + } + + let artifactsProcessed = 0; + let seriesStored = 0; + let samplesStored = 0; + let pointsLinked = 0; + let unmatchedArtifacts = 0; + let artifactFailures = 0; + let runFailures = 0; + let missingRuns = 0; + let goneRuns = 0; + + for (const [runIndex, run] of runs.entries()) { + const runId = run.github_run_id; + const repository = repositoryFromRunUrl(run.html_url) ?? DEFAULT_REPO; + const runStart = Date.now(); + const receiptPath = + flags.receipt ?? `power-publication-${runId}-attempt-${run.run_attempt}.json`; + const manifest: PowerPublicationManifest = fs.existsSync(receiptPath) + ? JSON.parse(fs.readFileSync(receiptPath, 'utf8')) + : { version: 1, runId, runAttempt: run.run_attempt, points: [] }; + if ( + manifest.version !== 1 || + manifest.runId !== runId || + manifest.runAttempt !== run.run_attempt || + !Array.isArray(manifest.points) || + (manifest.telemetry && + (manifest.telemetry.runId !== runId || manifest.telemetry.runAttempt !== run.run_attempt)) + ) + throw new Error('Publication receipt run/attempt does not match the recovery target'); + const saveReceipt = () => { + fs.mkdirSync(path.dirname(path.resolve(receiptPath)), { recursive: true }); + const temporary = `${receiptPath}.tmp`; + fs.writeFileSync(temporary, `${JSON.stringify(manifest, null, 2)}\n`, { flush: true }); + fs.renameSync(temporary, receiptPath); + }; + const refresh = async () => { + try { + await refreshBackfillBenchmarks(sql, manifest, saveReceipt); + } catch (error) { + process.exitCode = 1; + console.error( + ` Benchmark metadata refresh failed; retry --refresh-cache-only --run ${runId} --attempt ${run.run_attempt} --receipt ${receiptPath} --yes`, + error, + ); + } + }; + if (flags.refreshCacheOnly) { + await refresh(); + continue; + } + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `gpu-metrics-backfill-${runId}-`)); + const checkpointMetadata = async (updates: BenchmarkAuditUpdate[]) => { + if (updates.length === 0) return; + const candidates = await sql<{ id: number; offload_mode: string }[]>` + select br.id, br.offload_mode from benchmark_results br + where br.workflow_run_id = ${run.id} + and br.id = any(${sql.array(updates.map((update) => update.benchmarkResultId))}::bigint[]) + and br.benchmark_type = 'agentic_traces' and br.power_audit is null + `; + checkpointBenchmarkRefresh( + manifest, + candidates.map((row) => Number(row.id)), + saveReceipt, + candidates.map((row) => { + const update = updates.find((entry) => entry.benchmarkResultId === Number(row.id))!; + return { + ...update, + sourceIdentity: update.identity, + identity: { ...update.identity, offload_mode: row.offload_mode }, + }; + }), + ); + }; + const observations = new Map(); + const uniqueFallbacks = new Map(); + let recoveryError: string | undefined; + let expectationErrors: TelemetryReceipt['expectationErrors']; + try { + const listedArtifacts = await listBackfillRunArtifacts(repository, runId); + if (listedArtifacts === null) { + goneRuns++; + runFailures++; + recoveryError = `Run ${runId} attempt ${run.run_attempt} is gone from GitHub`; + console.log( + ` [${runIndex + 1}/${runs.length}] run ${runId} attempt ${run.run_attempt}: gone from GitHub`, + ); + continue; + } + const artifacts = telemetryArtifactsForAttempt( + listedArtifacts, + fetchRunMeta(repository, String(runId)), + run.run_attempt, + ); + const allPairs = pairGpuMetricsArtifacts(artifacts); + const pairs = allPairs.filter( + (pair) => !flags.artifact || pair.gpuMetrics.name === flags.artifact, + ); + // Expectations come from benchmark siblings even when no telemetry was + // uploaded. Counting only the successful pairs would hide missing points. + const missing = await collectMissingTelemetryExpectations( + artifacts, + allPairs, + flags.artifact, + async (artifact, onUnmapped) => { + let directory: string | null = null; + try { + directory = await retryArtifactOperation(`downloading ${artifact.name}`, () => + downloadArtifact(artifact, tempDir), + ); + const rows = await filterPurgedBenchmarkRows( + sql, + run, + readMappedBenchmarkRows(directory, onUnmapped), + ); + for (const row of rows) + await findBenchmarkResultIds(sql, run, [row], (id) => + uniqueFallbacks.set( + stablePowerPointIdentity(benchmarkPublicationIdentity(row)), + id, + ), + ); + return rows; + } finally { + if (directory) fs.rmSync(directory, { recursive: true, force: true }); + } + }, + ); + for (const observation of missing.observations) + observations.set(stablePowerPointIdentity(observation.identity), observation); + expectationErrors = missing.errors; + artifactFailures += new Set(missing.errors.map((error) => error.benchmarkArtifact)).size; + if (pairs.length === 0) { + if (flags.artifact) { + artifactFailures++; + recoveryError = `No retained telemetry pair for ${flags.artifact}`; + } + missingRuns++; + console.log( + ` [${runIndex + 1}/${runs.length}] run ${runId} attempt ${run.run_attempt}: no retained gpu_metrics pairs`, + ); + continue; + } + + const limiter = new AsyncSemaphore(flags.parallel); + const outcomes = await Promise.all( + pairs.map((pair) => + limiter.run(() => + processPair( + run, + pair, + tempDir, + observations, + missing.errors, + uniqueFallbacks, + checkpointMetadata, + ), + ), + ), + ); + let runSeries = 0; + let runSamples = 0; + for (const outcome of outcomes) { + switch (outcome.kind) { + case 'ingested': { + artifactsProcessed++; + if (outcome.expectationsUnknown) artifactFailures++; + runSeries += outcome.seriesCount; + runSamples += outcome.samplesInserted; + pointsLinked += outcome.pointsLinked; + checkpointBenchmarkRefresh( + manifest, + outcome.metadataUpdatedBenchmarkResultIds, + saveReceipt, + ); + break; + } + case 'unmatched': { + unmatchedArtifacts++; + break; + } + case 'failed': { + artifactFailures++; + break; + } + } + } + seriesStored += runSeries; + samplesStored += runSamples; + console.log( + ` [${runIndex + 1}/${runs.length}] run ${runId} attempt ${run.run_attempt} (${run.date}): ` + + `${pairs.length} pair(s), ${runSeries} series, +${runSamples} samples ` + + `(${((Date.now() - runStart) / 1000).toFixed(1)}s)`, + ); + } catch (error) { + runFailures++; + recoveryError = error instanceof Error ? error.message : String(error); + console.error(` ✗ run ${runId}:`, error); + } finally { + manifest.telemetry = await readTelemetryReceipt( + sql, + { runId, runAttempt: run.run_attempt }, + [...observations.values()], + { previous: manifest.telemetry, targeted: Boolean(flags.artifact), uniqueFallbacks }, + ); + if (recoveryError) { + const priorError = manifest.telemetry.recoveryError; + const priorScope = manifest.telemetry.recoveryArtifactNames; + manifest.telemetry.recoveryError = [ + ...new Set([priorError, recoveryError].filter(Boolean).join('\n').split('\n')), + ].join('\n'); + if (flags.artifact && (!priorError || priorScope)) + manifest.telemetry.recoveryArtifactNames = [ + ...new Set([...(priorScope ?? []), flags.artifact]), + ]; + else delete manifest.telemetry.recoveryArtifactNames; + } + if (expectationErrors) { + const failedBenchmarks = new Set(expectationErrors.map((error) => error.benchmarkArtifact)); + manifest.telemetry.expectationErrors = [ + ...(manifest.telemetry.expectationErrors ?? []).filter( + (error) => !failedBenchmarks.has(error.benchmarkArtifact), + ), + ...expectationErrors, + ]; + } + summarizeTelemetryReceipt(manifest.telemetry); + saveReceipt(); + if (manifest.benchmarkRefresh && manifest.benchmarkRefresh.status !== 'complete') + await refresh(); + console.log(` PowerX ingest receipt: ${receiptPath}`); + fs.rmSync(tempDir, { recursive: true, force: true }); + } + } + + console.log( + `\n=== backfill complete: ${artifactsProcessed} artifact(s), ${seriesStored} series, ` + + `${samplesStored} sample(s), ${pointsLinked} point link(s), ` + + `${unmatchedArtifacts} unmatched artifact(s), ` + + `${missingRuns} run(s) without pairs, ${goneRuns} run(s) gone from GitHub, ` + + `${artifactFailures} failed artifact(s), ${runFailures} failed run(s) ===`, + ); + console.log( + ' Point telemetry reads use the stored revision; API verification is recorded separately.', + ); + if (artifactFailures > 0 || runFailures > 0) process.exitCode = 1; +} + +runBackfillMain('backfill-gpu-metrics', sql, main); diff --git a/packages/db/src/backfill-server-log-files.ts b/packages/db/src/backfill-server-log-files.ts index 8047e0db5..46c0d05be 100644 --- a/packages/db/src/backfill-server-log-files.ts +++ b/packages/db/src/backfill-server-log-files.ts @@ -19,24 +19,25 @@ import os from 'node:os'; import path from 'node:path'; import { hasNoSslFlag } from './cli-utils.js'; -import { mapBenchmarkRow, type BenchmarkParams } from './etl/benchmark-mapper.js'; import { insertServerLogFilePaths } from './etl/benchmark-ingest.js'; import { createAdminSql } from './etl/db-utils.js'; import { listServerLogFilePaths, serverLogArtifactRoot } from './etl/server-log-artifacts.js'; -import { createSkipTracker } from './etl/skip-tracker.js'; -import { downloadArtifact, listRunArtifacts } from './lib/github-artifacts.js'; +import { downloadArtifact } from './lib/github-artifacts.js'; import { downloadGcsArtifact, listGcsServerLogArtifacts, type GcsArtifactMeta, } from './lib/gcs-artifacts.js'; -import { confirmProceed, parseLimitForceFlags, runBackfillMain } from './lib/backfill-runner.js'; +import { + confirmProceed, + listBackfillRunArtifacts, + parseLimitForceFlags, + runBackfillMain, +} from './lib/backfill-runner.js'; import { retryArtifactOperation } from './lib/artifact-retry.js'; +import { findBenchmarkResultIds, readMappedBenchmarkRows } from './lib/benchmark-result-lookup.js'; import { repositoryFromRunUrl } from './lib/runtime-metadata-artifacts.js'; -import { - pairServerLogArtifacts, - resolveServerLogResultCandidates, -} from './lib/server-log-backfill.js'; +import { pairServerLogArtifacts } from './lib/server-log-backfill.js'; const DEFAULT_REPO = 'SemiAnalysisAI/InferenceX'; const RETENTION_DAYS = 90; @@ -94,86 +95,6 @@ function parseBackfillFlags(): BackfillFlags { }; } -function findJsonFiles(root: string): string[] { - const files: string[] = []; - for (const entry of fs.readdirSync(root, { withFileTypes: true })) { - const pathname = path.join(root, entry.name); - if (entry.isDirectory()) files.push(...findJsonFiles(pathname)); - else if (entry.isFile() && entry.name.endsWith('.json')) files.push(pathname); - } - return files.toSorted(); -} - -function readMappedRows(root: string): BenchmarkParams[] { - const tracker = createSkipTracker(); - const rows: BenchmarkParams[] = []; - for (const file of findJsonFiles(root)) { - const parsed = JSON.parse(fs.readFileSync(file, 'utf8')) as unknown; - const rawRows = Array.isArray(parsed) ? parsed : [parsed]; - for (const raw of rawRows) { - if (!raw || typeof raw !== 'object' || Array.isArray(raw)) continue; - const mapped = mapBenchmarkRow(raw as Record, tracker); - if (mapped) rows.push(mapped); - } - } - return rows; -} - -async function findBenchmarkResultIds( - run: CandidateRun, - rows: readonly BenchmarkParams[], -): Promise { - const ids = new Set(); - for (const row of rows) { - const c = row.config; - const candidates = await sql<{ id: number; offload_mode: string }[]>` - select br.id, br.offload_mode - from benchmark_results br - join workflow_runs wr on wr.id = br.workflow_run_id - join configs cfg on cfg.id = br.config_id - where wr.github_run_id = ${run.github_run_id} - and wr.run_attempt = ${run.run_attempt} - and cfg.hardware = ${c.hardware} - and cfg.framework = ${c.framework} - and cfg.model = ${c.model} - and cfg.precision = ${c.precision} - and cfg.spec_method = ${c.specMethod} - and cfg.disagg = ${c.disagg} - and cfg.is_multinode = ${c.isMultinode} - and cfg.prefill_tp = ${c.prefillTp} - and cfg.prefill_ep = ${c.prefillEp} - and cfg.prefill_dp_attention = ${c.prefillDpAttn} - and cfg.prefill_num_workers = ${c.prefillNumWorkers} - and cfg.decode_tp = ${c.decodeTp} - and cfg.decode_ep = ${c.decodeEp} - and cfg.decode_dp_attention = ${c.decodeDpAttn} - and cfg.decode_num_workers = ${c.decodeNumWorkers} - and cfg.num_prefill_gpu = ${c.numPrefillGpu} - and cfg.num_decode_gpu = ${c.numDecodeGpu} - and br.benchmark_type = ${row.benchmarkType} - and br.isl is not distinct from ${row.isl} - and br.osl is not distinct from ${row.osl} - and br.conc = ${row.conc} - and br.recipe_fingerprint is not distinct from ${row.recipeFingerprint} - `; - const resolution = resolveServerLogResultCandidates( - candidates.map((candidate) => ({ - id: Number(candidate.id), - offloadMode: candidate.offload_mode, - })), - row.offloadMode, - ); - if (resolution.usedUniqueFallback) { - console.warn( - ` [WARN] benchmark result ${resolution.ids[0]} uses a historical offload label; ` + - `matched uniquely without offload_mode`, - ); - } - for (const id of resolution.ids) ids.add(id); - } - return [...ids]; -} - async function resultLogsAreComplete(resultIds: readonly number[]): Promise { if (resultIds.length === 0) return false; const [row] = await sql<{ complete: boolean }[]>` @@ -281,11 +202,8 @@ async function main(): Promise { flags.source !== 'gcs' && (flags.source === 'github' || isWithinGithubRetention(run.date)) ) { - const artifacts = await retryArtifactOperation( - `listing GitHub artifacts for run ${runId}`, - () => listRunArtifacts(repository, String(runId)), - ); - pairs = pairServerLogArtifacts(artifacts.filter((artifact) => !artifact.expired)); + const artifacts = await listBackfillRunArtifacts(repository, runId); + pairs = pairServerLogArtifacts((artifacts ?? []).filter((artifact) => !artifact.expired)); if (pairs.length > 0) { source = 'github'; githubRuns++; @@ -311,8 +229,13 @@ async function main(): Promise { : await retryArtifactOperation(`downloading ${pair.benchmarks.name}`, () => downloadArtifact(pair.benchmarks, tempDir), ); - const mappedRows = readMappedRows(benchmarkDir); - const resultIds = await findBenchmarkResultIds(run, mappedRows); + const mappedRows = readMappedBenchmarkRows(benchmarkDir); + const resultIds = await findBenchmarkResultIds(sql, run, mappedRows, (id) => + console.warn( + ` [WARN] benchmark result ${id} uses a historical offload label; ` + + `matched uniquely without offload_mode`, + ), + ); if (resultIds.length === 0) { unmatchedArtifacts++; console.warn(` [WARN] ${pair.serverLogs.name}: no matching benchmark rows`); diff --git a/packages/db/src/etl/__fixtures__/required-power-agentic-matrix.json b/packages/db/src/etl/__fixtures__/required-power-agentic-matrix.json deleted file mode 100644 index 4d17ed15a..000000000 --- a/packages/db/src/etl/__fixtures__/required-power-agentic-matrix.json +++ /dev/null @@ -1,50 +0,0 @@ -{ - "single_node": {}, - "multi_node": { - "agentic": [ - { - "image": "vllm/vllm-openai:nightly-dev-x86_64-cu13.0.1-728d3ad", - "model": "moonshotai/Kimi-K3", - "model-prefix": "kimik3", - "precision": "fp4", - "framework": "dynamo-vllm", - "spec-decoding": "mtp", - "runner": "cluster:b200-nscale", - "require-power": true, - "node-count": 2, - "prefill": { - "num-worker": 1, - "tp": 8, - "pp": 2, - "dcp-size": 8, - "pcp-size": 1, - "ep": 1, - "dp-attn": false, - "additional-settings": [ - "CONFIG_FILE=recipes/vllm/kimi-k3/agentic/agg-b200-tp8pp2-mooncake-c1-agentic.yaml", - "SYNTHETIC_ACCEPTANCE=true", - "SYNTHETIC_ACCEPTANCE_LENGTH=3.84" - ] - }, - "decode": { - "num-worker": 0, - "tp": 8, - "pp": 2, - "dcp-size": 8, - "pcp-size": 1, - "ep": 1, - "dp-attn": false, - "additional-settings": [] - }, - "conc": [1], - "kv-offloading": "none", - "total-cpu-dram-gb": 0, - "duration": 3600, - "exp-name": "kimik3_p1x8_d0x8_conc1", - "disagg": false, - "scenario-type": "agentic-coding", - "recipe-fingerprint": "ebb89391c58441525072eeffe36b6a6f6f52445116c68c99c017f06d662b972e" - } - ] - } -} diff --git a/packages/db/src/etl/benchmark-mapper.test.ts b/packages/db/src/etl/benchmark-mapper.test.ts index 962f5de16..5982348d4 100644 --- a/packages/db/src/etl/benchmark-mapper.test.ts +++ b/packages/db/src/etl/benchmark-mapper.test.ts @@ -1633,3 +1633,81 @@ describe('mapBenchmarkRow — v3 agentic nested agg schema', () => { expect(result!.offloadMode).toBe('on'); }); }); + +describe('NVL72 CPU ingestion', () => { + it.each([ + { gpu: 1, cpu: 1, cpuValid: 1, watts: 501 }, + { gpu: 0, cpu: '1', cpuValid: 1, watts: 501 }, + { gpu: 1, cpu: 0, cpuValid: 0, watts: undefined }, + { gpu: 1, cpu: 'invalid', cpuValid: 0, watts: undefined }, + { gpu: 1, cpu: undefined, cpuValid: undefined, watts: undefined }, + ])( + 'keeps CPU and GPU verdicts independent: GPU=$gpu CPU=$cpu', + ({ gpu, cpu, cpuValid, watts }) => { + const result = mapBenchmarkRow( + makeV2Row({ + power_valid: gpu, + ...(cpu === undefined ? {} : { cpu_power_valid: cpu }), + avg_power_w: 600, + avg_cpu_socket_power_w: 250.5, + avg_total_cpu_power_w: 501, + total_cpu_energy_j: 10020, + avg_total_module_power_w: 2901, + total_module_energy_j: 58020, + }), + createSkipTracker(), + ); + expect(result!.metrics.cpu_power_valid).toBe(cpuValid); + expect(result!.metrics.avg_total_cpu_power_w).toBe(watts); + expect(result!.metrics.avg_cpu_socket_power_w).toBe(watts === undefined ? undefined : 250.5); + expect(result!.metrics.total_cpu_energy_j).toBe(watts === undefined ? undefined : 10020); + expect(result!.metrics.avg_total_module_power_w).toBe(watts === undefined ? undefined : 2901); + expect(result!.metrics.total_module_energy_j).toBe(watts === undefined ? undefined : 58020); + expect(result!.metrics.avg_power_w).toBe(gpu === 1 ? 600 : undefined); + }, + ); + + it('preserves Grace socket evidence while bounding malformed CPU audit fields', () => { + const cpu = { + sensor_kind: 'grace_socket', + source: 'acpi', + expected_sockets: 4, + observed_sockets: 4, + sample_row_count: 6268, + reason_codes: [], + }; + const result = mapBenchmarkRow(makeV2Row({ power_audit: { cpu } }), createSkipTracker()); + expect(result!.powerAudit).toMatchObject({ cpu }); + expect( + extractPowerAudit({ + sample_count: 1, + cpu: { + sensor_kind: 'unknown', + source: 'x'.repeat(33), + expected_sockets: -1, + observed_sockets: Number.NaN, + sample_row_count: '12', + reason_codes: ['cpu_socket_count_mismatch', 'cpu_socket_count_mismatch', '', 7], + }, + }), + ).toEqual({ + sample_count: 1, + producer_sha: null, + exporter_image_sha256: null, + cpu: { sample_row_count: 12, reason_codes: ['cpu_socket_count_mismatch'] }, + }); + expect(extractPowerAudit({ sample_count: 1, cpu: 'acpi' })).not.toHaveProperty('cpu'); + }); + + it('scrubs supplemental CPU measurements after normalizing their verdict', () => { + const metrics = { + cpu_power_valid: 2, + power_valid: 1, + avg_total_cpu_power_w: 501, + avg_power_w: 600, + }; + normalizePowerContractMetrics(metrics, metrics); + expect(scrubWithheldPowerMetrics(metrics)).toBe(false); + expect(metrics).toEqual({ cpu_power_valid: 0, power_valid: 1, avg_power_w: 600 }); + }); +}); diff --git a/packages/db/src/etl/benchmark-mapper.ts b/packages/db/src/etl/benchmark-mapper.ts index 060b39412..3f3855778 100644 --- a/packages/db/src/etl/benchmark-mapper.ts +++ b/packages/db/src/etl/benchmark-mapper.ts @@ -7,6 +7,7 @@ import type { ConfigParams } from './config-cache'; import type { SkipTracker } from './skip-tracker'; import { + CPU_SIDE_POWER_METRIC_KEYS, MEASURED_POWER_METRIC_KEYS, METRIC_KEYS, PRECISION_KEYS, @@ -148,10 +149,28 @@ export interface PowerAudit { max_sample_gap_s?: number; producer_sha?: string | null; exporter_image_sha256?: string | null; - /** Relative path within the source run artifact bundle. */ + /** Validation filename; nested AgentX documents use a canonical filename alias. */ source?: string; /** Producer device identifiers; not necessarily physical UUIDs on older traces. */ observed_gpu_ids?: string[]; + /** NVL72 CPU-side leg provenance; present only when the producer ran that leg. */ + cpu?: PowerAuditCpu; +} + +/** Sensor kinds the consumer's CPU-side leg can select as the headline series. */ +export const POWER_AUDIT_CPU_SENSOR_KINDS = ['module', 'grace_socket', 'dcgm_cpu_rail'] as const; + +/** + * Bounded CPU-side provenance emitted next to `cpu_power_valid`: which sensor fed + * the Grace-side keys, its source, socket coverage, and the leg's reason codes. + */ +export interface PowerAuditCpu { + sensor_kind?: (typeof POWER_AUDIT_CPU_SENSOR_KINDS)[number]; + source?: string; + expected_sockets?: number; + observed_sockets?: number; + sample_row_count?: number; + reason_codes?: string[]; } export interface BenchmarkParams { @@ -540,11 +559,13 @@ export function normalizePowerContractMetrics( row: Record, metrics: Record, ): void { - if (Object.hasOwn(row, 'power_valid')) { - const verdict = row.power_valid; - metrics.power_valid = verdict === 1 || verdict === '1' ? 1 : 0; - } else { - delete metrics.power_valid; + for (const field of ['power_valid', 'cpu_power_valid'] as const) { + if (Object.hasOwn(row, field)) { + const verdict = row[field]; + metrics[field] = verdict === 1 || verdict === '1' ? 1 : 0; + } else { + delete metrics[field]; + } } if (!Object.hasOwn(row, 'power_metric_schema_version')) { @@ -569,15 +590,24 @@ export function normalizePowerContractMetrics( /** * Enforces fail-closed power publication at ingest. An explicit normalized - * invalid verdict removes every measured field while preserving the contract - * and diagnostic fields; legacy rows without a verdict remain unchanged. - * Returns true so callers also drop worker telemetry. Paths that bypass - * `mapBenchmarkRow` must normalize the verdict before calling this function. - * Queries intentionally remain raw; the frontend withholds independently. + * invalid GPU verdict removes every GPU-side measured field while preserving + * the contract and diagnostic fields; legacy rows without a verdict remain + * unchanged. The NVL72 CPU-side keys follow their own `cpu_power_valid` + * verdict instead: anything but a normalized 1 withholds them, because no + * legacy rows predate that verdict and the estimator admits rows the same way. + * Returns true when GPU power was withheld so callers also drop worker + * telemetry. Paths that bypass `mapBenchmarkRow` must normalize the verdicts + * before calling this function. Queries intentionally remain raw; the + * frontend withholds independently. */ export function scrubWithheldPowerMetrics(metrics: Record): boolean { + if (metrics.cpu_power_valid !== 1) { + for (const key of CPU_SIDE_POWER_METRIC_KEYS) delete metrics[key]; + } if (metrics.power_valid !== 0) return false; - for (const key of MEASURED_POWER_METRIC_KEYS) delete metrics[key]; + for (const key of MEASURED_POWER_METRIC_KEYS) { + if (!CPU_SIDE_POWER_METRIC_KEYS.has(key)) delete metrics[key]; + } return true; } @@ -666,6 +696,32 @@ function auditSha(v: unknown): string | null { return s.length > 0 && s.length <= MAX_POWER_AUDIT_SHA_LENGTH ? s : null; } +/** + * Narrow the CPU-side leg's provenance. Reason codes reuse the producer reason + * grammar; an unrecognised sensor kind is dropped rather than stored as a label + * the dashboard would misread. Undefined when nothing well-formed remains. + */ +function extractPowerAuditCpu(raw: unknown): PowerAuditCpu | undefined { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) return undefined; + const e = raw as Record; + const cpu: PowerAuditCpu = {}; + if ((POWER_AUDIT_CPU_SENSOR_KINDS as readonly unknown[]).includes(e.sensor_kind)) { + cpu.sensor_kind = e.sensor_kind as PowerAuditCpu['sensor_kind']; + } + if (typeof e.source === 'string' && e.source.length > 0 && e.source.length <= 32) { + cpu.source = e.source; + } + for (const field of ['expected_sockets', 'observed_sockets', 'sample_row_count'] as const) { + const n = auditCount(e[field]); + if (n !== undefined) cpu[field] = n; + } + // An empty list is the producer's "leg valid, nothing to report"; keep it. + if (Array.isArray(e.reason_codes)) { + cpu.reason_codes = extractPowerInvalidReasons(e.reason_codes) ?? []; + } + return Object.keys(cpu).length > 0 ? cpu : undefined; +} + /** * Missing or malformed audit values must not become a fabricated measurement; * SQL NULL distinguishes absent evidence from an empty recorded object. @@ -701,6 +757,8 @@ export function extractPowerAudit(raw: unknown): PowerAudit | undefined { ); if (ids.length > 0) audit.observed_gpu_ids = [...new Set(ids)].slice(0, 1024); } + const cpu = extractPowerAuditCpu(e.cpu); + if (cpu !== undefined) audit.cpu = cpu; const hasNumericField = Object.keys(audit).length > 0; audit.producer_sha = auditSha(e.producer_sha); diff --git a/packages/db/src/etl/gpu-metrics-artifacts.ts b/packages/db/src/etl/gpu-metrics-artifacts.ts new file mode 100644 index 000000000..d101976e4 --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-artifacts.ts @@ -0,0 +1,275 @@ +/** + * Filesystem discovery for `gpu_metrics_` artifacts. + * + * Layout uploaded by the producer (`benchmark-tmpl.yml` "Upload GPU metrics"): + * gpu_metrics.csv fixed-sequence jobs + * results/gpu_metrics.csv AgentX jobs (per-concurrency) + * results/gpu_metrics*.csv multinode staging, one CSV per node + * gpu_metrics*_context.json collector context (timestamp zone …) + * gpu_metrics_identity.{json,csv} SKU / UUID / driver per GPU + * gpu_metrics_energy_{start,end}.csv amd-smi energy counters + * + * Multinode jobs (`benchmark-multinode-tmpl.yml`) upload no `gpu_metrics_` + * artifact; their telemetry travels inside `power_audit_`: + * LOGS/power/samples.csv deployment-wide DCGM power, one row per (host, GPU, s) + * LOGS/power/manifest.json producer, source metric, cadence + * Single-node jobs upload a `power_audit_` bundle too, so it only stands in + * for a point when no `gpu_metrics_` sibling exists. + * + * `eval_gpu_metrics_` (eval-only jobs) is deliberately ignored: those + * jobs produce no benchmark point to attach the telemetry to. + */ + +import fs from 'node:fs'; +import path from 'node:path'; +import { createHash } from 'node:crypto'; + +import { isMultinodePowerSamplesPath } from './multinode-power-samples.js'; +import { + isPowerAuditValidationEntry, + normalizePowerAuditValidations, +} from './power-audit-validations.js'; + +export const GPU_METRICS_ARTIFACT_PREFIX = 'gpu_metrics_'; +export const POWER_AUDIT_ARTIFACT_PREFIX = 'power_audit_'; + +export interface GpuMetricsArtifact { + artifactName: string; + artifactDir: string; +} + +export interface GpuMetricsCsvFile { + /** POSIX-style path relative to the artifact root. */ + fileName: string; + path: string; +} + +export interface GpuMetricsSidecars { + context: Record | null; + /** Audit documents keyed by canonical source; nested documents retain original path/hash. */ + validations?: Record>; + /** Expected stored files and deduplicated row counts from this one artifact. */ + seriesInventory?: { fileName: string; sampleCount: number }[]; + /** Bundle manifest when the preferred CSV has its own collector context. */ + powerManifest?: Record | null; + identity: unknown | null; + energyStart: Record | null; + energyEnd: Record | null; +} + +/** Return the shared suffix that pairs a telemetry artifact with bmk[_agentic]_. */ +export function gpuMetricsArtifactSuffix(artifactName: string): string | null { + for (const prefix of [GPU_METRICS_ARTIFACT_PREFIX, POWER_AUDIT_ARTIFACT_PREFIX]) { + if (artifactName.startsWith(prefix)) return artifactName.slice(prefix.length); + } + return null; +} + +/** The nvidia-smi/amd-smi CSV carries clocks, utilization and temperature; the power bundle only power. */ +export function isPowerAuditArtifact(artifactName: string): boolean { + return artifactName.startsWith(POWER_AUDIT_ARTIFACT_PREFIX); +} + +function isGpuMetricsCsvName(fileName: string): boolean { + const lower = fileName.toLowerCase(); + if (!lower.startsWith('gpu_metrics') || !lower.endsWith('.csv')) return false; + // Sidecars share the prefix but are not time series. + return !lower.includes('_identity') && !lower.includes('_energy_'); +} + +/** Recursively list every telemetry CSV under an extracted artifact root. */ +export function listGpuMetricsCsvFiles(root: string): GpuMetricsCsvFile[] { + if (!fs.existsSync(root) || !fs.statSync(root).isDirectory()) return []; + const files: GpuMetricsCsvFile[] = []; + const visit = (directory: string): void => { + for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { + const pathname = path.join(directory, entry.name); + if (entry.isDirectory()) visit(pathname); + else if (entry.isFile() && isGpuMetricsCsvName(entry.name)) { + files.push({ + fileName: path.relative(root, pathname).split(path.sep).join('/'), + path: pathname, + }); + } + } + }; + visit(root); + return files.toSorted((a, b) => a.fileName.localeCompare(b.fileName)); +} + +/** Every multinode power CSV under an extracted `power_audit_` root. */ +export function listMultinodePowerSampleFiles(root: string): GpuMetricsCsvFile[] { + if (!fs.existsSync(root) || !fs.statSync(root).isDirectory()) return []; + const files: GpuMetricsCsvFile[] = []; + const visit = (directory: string): void => { + for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { + const pathname = path.join(directory, entry.name); + if (entry.isDirectory()) visit(pathname); + else if (entry.isFile()) { + const fileName = path.relative(root, pathname).split(path.sep).join('/'); + if (isMultinodePowerSamplesPath(fileName)) files.push({ fileName, path: pathname }); + } + } + }; + visit(root); + return files.toSorted((a, b) => a.fileName.localeCompare(b.fileName)); +} + +/** The producer manifest next to `samples.csv`; null when absent or malformed. */ +export function readMultinodePowerManifest(samplesPath: string): Record | null { + const parsed = readJsonIfPresent(path.join(path.dirname(samplesPath), 'manifest.json')); + return parsed && typeof parsed === 'object' && !Array.isArray(parsed) + ? (parsed as Record) + : null; +} + +/** Preserve window boundaries and role overrides before GitHub artifact expiry. */ +export function readPowerAuditValidations( + root: string, + artifactName = path.basename(root), +): Record> { + if (!fs.existsSync(root) || !fs.statSync(root).isDirectory()) return {}; + const files = new Map(); + const read = (name: string) => { + const file = path.join(root, name); + if (isPowerAuditValidationEntry(name) && fs.existsSync(file) && fs.statSync(file).isFile()) { + files.set(name, fs.readFileSync(file, 'utf8')); + } + }; + for (const name of fs.readdirSync(root).sort()) { + read(name); + } + const agentic = path.join(root, 'LOGS', 'agentic'); + if (fs.existsSync(agentic) && fs.statSync(agentic).isDirectory()) { + for (const entry of fs.readdirSync(agentic, { withFileTypes: true })) { + const match = /^conc_(?[1-9]\d*)$/u.exec(entry.name); + if (!entry.isDirectory() || !match) continue; + read(`LOGS/agentic/${entry.name}/power_validation.json`); + read(`LOGS/power/windows/agentic_power_concurrency_${match.groups!.concurrency}.json`); + } + } + const validations = normalizePowerAuditValidations(artifactName, files); + for (const validation of validations.values()) { + if (typeof validation.validation_path !== 'string') continue; + const original = files.get(validation.validation_path); + if (original !== undefined) + validation.validation_sha256 = createHash('sha256').update(original).digest('hex'); + } + return Object.fromEntries(validations); +} + +function readJsonIfPresent(pathname: string): unknown | null { + if (!fs.existsSync(pathname)) return null; + try { + return JSON.parse(fs.readFileSync(pathname, 'utf8')) as unknown; + } catch { + return null; + } +} + +/** `gpu,total_energy_consumption` two-column CSV → { "": joules }. */ +export function parseEnergyCsv(text: string): Record | null { + const lines = text + .split('\n') + .map((line) => line.trim()) + .filter((line) => line.length > 0); + if (lines.length <= 1) return null; + const out: Record = {}; + for (const line of lines.slice(1)) { + const [gpu, value] = line.split(','); + const parsed = Number.parseFloat(value ?? ''); + if (gpu !== undefined && gpu !== '' && Number.isFinite(parsed)) out[gpu.trim()] = parsed; + } + return Object.keys(out).length > 0 ? out : null; +} + +/** Identity CSV (nvidia-smi) → array of column objects; JSON identity passes through. */ +function readIdentity(csvDir: string): unknown | null { + const json = readJsonIfPresent(path.join(csvDir, 'gpu_metrics_identity.json')); + if (json !== null) return json; + const csvPath = path.join(csvDir, 'gpu_metrics_identity.csv'); + if (!fs.existsSync(csvPath)) return null; + const lines = fs + .readFileSync(csvPath, 'utf8') + .split('\n') + .map((line) => line.trim()) + .filter((line) => line.length > 0); + if (lines.length <= 1) return null; + const header = lines[0]!.split(',').map((h) => h.trim()); + return lines.slice(1).map((line) => { + const cells = line.split(',').map((c) => c.trim()); + return Object.fromEntries(header.map((key, i) => [key, cells[i] ?? ''])); + }); +} + +/** + * Sidecars live next to the CSV they describe. The context file is either + * `gpu_metrics_context.json` or `_gpu_metrics_context.json`. + */ +export function readGpuMetricsSidecars(csvPath: string): GpuMetricsSidecars { + const dir = path.dirname(csvPath); + let context: Record | null = null; + // Match live ZIP reads even when filesystem/central-directory orders differ. + const contextFiles = fs + .readdirSync(dir) + .filter( + (entry) => entry.toLowerCase().endsWith('_context.json') && entry.includes('gpu_metrics'), + ) + .sort(); + for (const entry of contextFiles) { + const parsed = readJsonIfPresent(path.join(dir, entry)); + if (parsed && typeof parsed === 'object' && !Array.isArray(parsed)) { + context = parsed as Record; + break; + } + } + const energyStartPath = path.join(dir, 'gpu_metrics_energy_start.csv'); + const energyEndPath = path.join(dir, 'gpu_metrics_energy_end.csv'); + return { + context, + identity: readIdentity(dir), + energyStart: fs.existsSync(energyStartPath) + ? parseEnergyCsv(fs.readFileSync(energyStartPath, 'utf8')) + : null, + energyEnd: fs.existsSync(energyEndPath) + ? parseEnergyCsv(fs.readFileSync(energyEndPath, 'utf8')) + : null, + }; +} + +/** + * Collector clock offset in minutes east of UTC, from the context sidecar. + * The producer writes `{"timestamp_timezone":"UTC"}`; anything else that is + * not a fixed `±HH:MM` offset is treated as UTC because nvidia-smi timestamps + * carry no zone of their own. + */ +export function contextUtcOffsetMinutes(context: Record | null): number { + const zone = context?.timestamp_timezone; + if (typeof zone !== 'string') return 0; + const match = /^(?[+-])(?\d{2}):?(?\d{2})$/u.exec(zone.trim()); + if (!match?.groups) return 0; + const minutes = Number(match.groups.h) * 60 + Number(match.groups.m); + return match.groups.sign === '-' ? -minutes : minutes; +} + +/** + * Index every extracted telemetry artifact by its shared suffix. A + * `gpu_metrics_` upload wins over the `power_audit_` bundle for the same + * suffix; the bundle only fills in for multinode jobs that have no other. + */ +export function discoverGpuMetricsArtifacts(artifactsDir: string): Map { + const discovered = new Map(); + if (!fs.existsSync(artifactsDir)) return discovered; + const names = fs + .readdirSync(artifactsDir) + .filter((artifactName) => fs.statSync(path.join(artifactsDir, artifactName)).isDirectory()); + for (const preferred of [false, true]) { + for (const artifactName of names) { + if (isPowerAuditArtifact(artifactName) !== preferred) continue; + const suffix = gpuMetricsArtifactSuffix(artifactName); + if (!suffix || discovered.has(suffix)) continue; + discovered.set(suffix, { artifactName, artifactDir: path.join(artifactsDir, artifactName) }); + } + } + return discovered; +} diff --git a/packages/db/src/etl/gpu-metrics-csv.test.ts b/packages/db/src/etl/gpu-metrics-csv.test.ts new file mode 100644 index 000000000..be0425149 --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-csv.test.ts @@ -0,0 +1,84 @@ +import { describe, expect, it } from 'vitest'; + +import { computeGpuMetricStats, parseGpuMetricsCsv } from './gpu-metrics-csv.js'; + +const NVIDIA_CSV = [ + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]', + '2026/09/11 04:19:41.982, 0, 187.80 W, 33, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:41.986, 1, 190.96 W, 39, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:42.990, 0, 912.10 W, 61, 1965 MHz, 3996 MHz, 98 %, 74 %', + '2026/09/11 04:19:42.994, 1, [N/A], 62, 1965 MHz, 3996 MHz, 97 %, 73 %', + '2026/09/11 04:19:43.990, 0, 930.00 W, 63, 1965 MHz, 3996 MHz, 99 %, 75 %', +].join('\n'); + +const AMD_HEADER = + 'timestamp,gpu,gfx_activity,umc_activity,mm_activity,socket_power,gfx_voltage,soc_voltage,mem_voltage,gfx_0_clk,mem_0_clk,fclk_0_clk,socclk_0_clk,edge,hotspot,mem'; + +const AMD_CSV = [ + AMD_HEADER, + `1789515524,0,0,0,N/A,288,N/A,N/A,N/A,2402,2000,1250,39,N/A,44,24`, + `1789515525,0,95,80,N/A,1180,850,900,1250,2100,2000,1250,39,55,78,60`, + `1789515524,1,0,0,N/A,290,N/A,N/A,N/A,2402,2000,1250,39,40,45,25`, +].join('\r\n'); + +describe('parseGpuMetricsCsv — NVIDIA', () => { + it('parses unit-suffixed cells into UTC epoch samples and drops rows without power', () => { + const parsed = parseGpuMetricsCsv(NVIDIA_CSV); + expect(parsed?.vendor).toBe('nvidia'); + expect(parsed?.samples).toHaveLength(4); + const first = parsed!.samples[0]!; + expect(first.timestampMs).toBe(Date.UTC(2026, 8, 11, 4, 19, 41, 982)); + expect(first.gpuIndex).toBe(0); + expect(first.powerW).toBe(187.8); + expect(first.temperatureC).toBe(33); + expect(first.smClockMhz).toBe(120); + expect(first.memClockMhz).toBe(3996); + expect(first.gpuUtilPct).toBe(0); + expect(first.memUtilPct).toBe(0); + expect(first.edgeTempC).toBeNull(); + // The `[N/A]` power row for GPU 1 is not a usable sample. + expect(parsed!.samples.filter((s) => s.gpuIndex === 1)).toHaveLength(1); + }); +}); + +describe('parseGpuMetricsCsv — AMD', () => { + it('maps the amd-smi subset, preferring hotspot over edge temperature', () => { + const parsed = parseGpuMetricsCsv(AMD_CSV); + expect(parsed?.vendor).toBe('amd'); + expect(parsed?.samples).toHaveLength(3); + const idle = parsed!.samples[0]!; + expect(idle.timestampMs).toBe(1789515524000); + expect(idle.powerW).toBe(288); + expect(idle.temperatureC).toBe(44); + expect(idle.edgeTempC).toBeNull(); + expect(idle.memTempC).toBe(24); + expect(idle.gfxVoltageMv).toBeNull(); + const busy = parsed!.samples[1]!; + expect(busy.gpuUtilPct).toBe(95); + expect(busy.memUtilPct).toBe(80); + expect(busy.gfxVoltageMv).toBe(850); + expect(busy.fclkMhz).toBe(1250); + expect(busy.socclkMhz).toBe(39); + expect(busy.edgeTempC).toBe(55); + expect(busy.temperatureC).toBe(78); + }); +}); + +describe('computeGpuMetricStats', () => { + it('digests every non-null metric per GPU with interpolated percentiles', () => { + const parsed = parseGpuMetricsCsv(NVIDIA_CSV)!; + const stats = computeGpuMetricStats(parsed.samples); + const gpu0Power = stats.find((s) => s.gpuIndex === 0 && s.metric === 'powerW')!; + expect(gpu0Power.count).toBe(3); + expect(gpu0Power.min).toBe(187.8); + expect(gpu0Power.max).toBe(930); + expect(gpu0Power.mean).toBeCloseTo((187.8 + 912.1 + 930) / 3, 6); + expect(gpu0Power.median).toBe(912.1); + expect(gpu0Power.p95).toBeCloseTo(912.1 + (930 - 912.1) * 0.9, 6); + expect(gpu0Power.p99).toBeCloseTo(912.1 + (930 - 912.1) * 0.98, 6); + expect(gpu0Power.stddev).toBeGreaterThan(0); + // AMD-only columns never appear for an NVIDIA series. + expect(stats.some((s) => s.metric === 'edgeTempC')).toBe(false); + expect(stats.filter((s) => s.gpuIndex === 1 && s.metric === 'powerW')[0]!.count).toBe(1); + }); +}); diff --git a/packages/db/src/etl/gpu-metrics-csv.ts b/packages/db/src/etl/gpu-metrics-csv.ts new file mode 100644 index 000000000..35ec33e57 --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-csv.ts @@ -0,0 +1,385 @@ +/** + * Parser and digest for the producer's `gpu_metrics.csv` telemetry. + * + * The InferenceX runner samples `nvidia-smi --query-gpu=…` or `amd-smi metric + * --csv` once per second for the lifetime of a benchmark job. This module is + * pure (no I/O) so the CI ingest, the historical backfill, and the app's + * GitHub fallback path all normalize the two vendor formats identically. + */ + +export type GpuMetricsVendor = 'nvidia' | 'amd'; + +export interface GpuMetricSample { + /** Sample instant in epoch milliseconds (UTC). */ + timestampMs: number; + gpuIndex: number; + powerW: number | null; + temperatureC: number | null; + smClockMhz: number | null; + memClockMhz: number | null; + gpuUtilPct: number | null; + memUtilPct: number | null; + edgeTempC: number | null; + memTempC: number | null; + gfxVoltageMv: number | null; + socVoltageMv: number | null; + memVoltageMv: number | null; + fclkMhz: number | null; + socclkMhz: number | null; + mmActivityPct: number | null; +} + +export interface ParsedGpuMetricsCsv { + vendor: GpuMetricsVendor; + samples: GpuMetricSample[]; +} + +/** Sample columns that participate in the per-GPU statistics digest. */ +export const GPU_METRIC_STAT_KEYS = [ + 'powerW', + 'temperatureC', + 'smClockMhz', + 'memClockMhz', + 'gpuUtilPct', + 'memUtilPct', + 'edgeTempC', + 'memTempC', + 'gfxVoltageMv', + 'socVoltageMv', + 'memVoltageMv', + 'fclkMhz', + 'socclkMhz', + 'mmActivityPct', +] as const; + +export type GpuMetricStatKey = (typeof GPU_METRIC_STAT_KEYS)[number]; + +export interface GpuMetricStats { + gpuIndex: number; + metric: GpuMetricStatKey; + count: number; + min: number; + max: number; + mean: number; + median: number; + p95: number; + p99: number; + stddev: number; +} + +export function splitCsvLine(line: string): string[] { + const result: string[] = []; + let current = ''; + let inQuotes = false; + for (const char of line) { + if (char === '"') { + inQuotes = !inQuotes; + } else if (char === ',' && !inQuotes) { + result.push(current.trim()); + current = ''; + } else { + current += char; + } + } + result.push(current.trim()); + return result; +} + +function buildColumnMap(headerLine: string): Map { + const headers = splitCsvLine(headerLine); + const map = new Map(); + for (let i = 0; i < headers.length; i++) map.set(headers[i]!.toLowerCase(), i); + return map; +} + +/** `N/A`, empty, and unparseable cells become null; unit suffixes are ignored. */ +export function parseMetricCell(value: string | undefined): number | null { + if (value === undefined) return null; + const trimmed = value.trim(); + if (trimmed === '' || trimmed === 'N/A' || trimmed === '[N/A]') return null; + const parsed = Number.parseFloat(trimmed); + return Number.isFinite(parsed) ? parsed : null; +} + +const NVIDIA_TIMESTAMP_RE = + /^(?\d{4})\/(?\d{2})\/(?\d{2}) (?\d{2}):(?\d{2}):(?\d{2})(?:\.(?\d{1,3}))?$/u; + +/** + * nvidia-smi prints `YYYY/MM/DD HH:MM:SS.mmm` in the collector's local zone. + * The producer runs its collectors with TZ=UTC; callers can pass a different + * fixed offset (minutes east of UTC) when a context sidecar says otherwise. + */ +export function parseNvidiaTimestamp(raw: string, offsetMinutes = 0): number | null { + const match = NVIDIA_TIMESTAMP_RE.exec(raw.trim()); + if (!match?.groups) return null; + const { y, mo, d, h, mi, s, ms } = match.groups; + const millis = ms ? Number(ms.padEnd(3, '0')) : 0; + const utc = Date.UTC( + Number(y), + Number(mo) - 1, + Number(d), + Number(h), + Number(mi), + Number(s), + millis, + ); + return utc - offsetMinutes * 60_000; +} + +/** amd-smi prints Unix epoch seconds; accept fractional and millisecond forms. */ +export function parseAmdTimestamp(raw: string): number | null { + const trimmed = raw.trim(); + if (/^\d+(?:\.\d+)?$/u.test(trimmed)) { + const numeric = Number.parseFloat(trimmed); + if (numeric > 1e12) return Math.round(numeric); // already milliseconds + if (numeric > 1e9) return Math.round(numeric * 1000); + return null; + } + const iso = Date.parse(trimmed); + return Number.isFinite(iso) ? iso : null; +} + +function emptySample(timestampMs: number, gpuIndex: number): GpuMetricSample { + return { + timestampMs, + gpuIndex, + powerW: null, + temperatureC: null, + smClockMhz: null, + memClockMhz: null, + gpuUtilPct: null, + memUtilPct: null, + edgeTempC: null, + memTempC: null, + gfxVoltageMv: null, + socVoltageMv: null, + memVoltageMv: null, + fclkMhz: null, + socclkMhz: null, + mmActivityPct: null, + }; +} + +/** + * Columns: timestamp, index, power.draw [W], temperature.gpu, + * clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], + * utilization.memory [%]. Header order is fixed by NVIDIA_GPU_MONITOR_QUERY in + * the producer, but we still resolve by name so a reordered query keeps working. + */ +function parseNvidiaCsv(lines: readonly string[], offsetMinutes: number): GpuMetricSample[] { + const header = splitCsvLine(lines[0]!).map((h) => h.toLowerCase().replace(/\s*\[.*\]$/u, '')); + const col = (name: string) => header.indexOf(name); + const iTimestamp = col('timestamp'); + const iIndex = col('index'); + const iPower = col('power.draw'); + const iTemp = col('temperature.gpu'); + const iSm = col('clocks.current.sm'); + const iMem = col('clocks.current.memory'); + const iUtil = col('utilization.gpu'); + const iMemUtil = col('utilization.memory'); + if (iTimestamp < 0 || iIndex < 0 || iPower < 0) return []; + + const samples: GpuMetricSample[] = []; + for (let i = 1; i < lines.length; i++) { + const cols = splitCsvLine(lines[i]!); + if (cols.length <= Math.max(iTimestamp, iIndex, iPower)) continue; + const timestampMs = parseNvidiaTimestamp(cols[iTimestamp]!, offsetMinutes); + const gpuIndex = Number.parseInt(cols[iIndex]!, 10); + if (timestampMs === null || !Number.isInteger(gpuIndex) || gpuIndex < 0) continue; + const sample = emptySample(timestampMs, gpuIndex); + sample.powerW = parseMetricCell(cols[iPower]); + if (sample.powerW === null) continue; + sample.temperatureC = iTemp >= 0 ? parseMetricCell(cols[iTemp]) : null; + sample.smClockMhz = iSm >= 0 ? parseMetricCell(cols[iSm]) : null; + sample.memClockMhz = iMem >= 0 ? parseMetricCell(cols[iMem]) : null; + sample.gpuUtilPct = iUtil >= 0 ? parseMetricCell(cols[iUtil]) : null; + sample.memUtilPct = iMemUtil >= 0 ? parseMetricCell(cols[iMemUtil]) : null; + samples.push(sample); + } + return samples; +} + +/** + * amd-smi `metric --csv` is a wide table. The consumed subset mirrors the + * dashboard: socket_power, gfx_activity, umc_activity, mm_activity, the first + * gfx/mem/fclk/socclk clock domains, hotspot/edge/mem temperatures and the + * three voltage rails. + */ +function parseAmdCsv(lines: readonly string[]): GpuMetricSample[] { + const colMap = buildColumnMap(lines[0]!); + const col = (name: string) => colMap.get(name) ?? -1; + const iTimestamp = col('timestamp'); + const iGpu = col('gpu'); + const iPower = col('socket_power'); + const iGfxActivity = col('gfx_activity'); + const iUmcActivity = col('umc_activity'); + const iGfxClk = col('gfx_0_clk'); + const iMemClk = col('mem_0_clk'); + const iHotspot = col('hotspot'); + const iEdge = col('edge'); + const iMemTemp = col('mem'); + const iGfxVoltage = col('gfx_voltage'); + const iSocVoltage = col('soc_voltage'); + const iMemVoltage = col('mem_voltage'); + const iFclk = col('fclk_0_clk'); + const iSocClk = col('socclk_0_clk'); + const iMmActivity = col('mm_activity'); + if (iTimestamp < 0 || iGpu < 0 || iPower < 0) return []; + + const samples: GpuMetricSample[] = []; + for (let i = 1; i < lines.length; i++) { + const cols = splitCsvLine(lines[i]!); + if (cols.length <= Math.max(iTimestamp, iGpu, iPower)) continue; + const timestampMs = parseAmdTimestamp(cols[iTimestamp]!); + const gpuIndex = Number.parseInt(cols[iGpu]!, 10); + if (timestampMs === null || !Number.isInteger(gpuIndex) || gpuIndex < 0) continue; + const sample = emptySample(timestampMs, gpuIndex); + sample.powerW = parseMetricCell(cols[iPower]); + if (sample.powerW === null) continue; + const hotspot = iHotspot >= 0 ? parseMetricCell(cols[iHotspot]) : null; + const edge = iEdge >= 0 ? parseMetricCell(cols[iEdge]) : null; + sample.temperatureC = hotspot ?? edge; + sample.edgeTempC = edge; + sample.memTempC = iMemTemp >= 0 ? parseMetricCell(cols[iMemTemp]) : null; + sample.smClockMhz = iGfxClk >= 0 ? parseMetricCell(cols[iGfxClk]) : null; + sample.memClockMhz = iMemClk >= 0 ? parseMetricCell(cols[iMemClk]) : null; + sample.gpuUtilPct = iGfxActivity >= 0 ? parseMetricCell(cols[iGfxActivity]) : null; + sample.memUtilPct = iUmcActivity >= 0 ? parseMetricCell(cols[iUmcActivity]) : null; + sample.gfxVoltageMv = iGfxVoltage >= 0 ? parseMetricCell(cols[iGfxVoltage]) : null; + sample.socVoltageMv = iSocVoltage >= 0 ? parseMetricCell(cols[iSocVoltage]) : null; + sample.memVoltageMv = iMemVoltage >= 0 ? parseMetricCell(cols[iMemVoltage]) : null; + sample.fclkMhz = iFclk >= 0 ? parseMetricCell(cols[iFclk]) : null; + sample.socclkMhz = iSocClk >= 0 ? parseMetricCell(cols[iSocClk]) : null; + sample.mmActivityPct = iMmActivity >= 0 ? parseMetricCell(cols[iMmActivity]) : null; + samples.push(sample); + } + return samples; +} + +export interface ParseGpuMetricsOptions { + /** Fixed offset of the NVIDIA collector clock, minutes east of UTC. */ + nvidiaUtcOffsetMinutes?: number; +} + +/** + * Parse one gpu_metrics CSV, auto-detecting the vendor from the header. + * Samples are returned in file order; callers sort per GPU as needed. + */ +export function parseGpuMetricsCsv( + csvText: string, + options: ParseGpuMetricsOptions = {}, +): ParsedGpuMetricsCsv | null { + const lines = csvText + .split('\n') + .map((line) => line.replace(/\r$/u, '')) + .filter((line) => line.trim().length > 0); + if (lines.length <= 1) return null; + const headerLower = lines[0]!.toLowerCase(); + if (headerLower.includes('socket_power') || headerLower.includes('gfx_activity')) { + return { vendor: 'amd', samples: parseAmdCsv(lines) }; + } + if (headerLower.includes('power.draw')) { + return { + vendor: 'nvidia', + samples: parseNvidiaCsv(lines, options.nvidiaUtcOffsetMinutes ?? 0), + }; + } + return null; +} + +function percentile(sorted: readonly number[], p: number): number { + const idx = (p / 100) * (sorted.length - 1); + const lo = Math.floor(idx); + const hi = Math.ceil(idx); + if (lo === hi) return sorted[lo]!; + return sorted[lo]! + (sorted[hi]! - sorted[lo]!) * (idx - lo); +} + +/** Per-GPU, per-metric summary over every non-null sample. */ +export function computeGpuMetricStats(samples: readonly GpuMetricSample[]): GpuMetricStats[] { + const groups = new Map< + string, + { gpuIndex: number; metric: GpuMetricStatKey; values: number[] } + >(); + for (const sample of samples) { + for (const metric of GPU_METRIC_STAT_KEYS) { + const value = sample[metric]; + if (value === null) continue; + const key = `${sample.gpuIndex}|${metric}`; + let group = groups.get(key); + if (!group) { + group = { gpuIndex: sample.gpuIndex, metric, values: [] }; + groups.set(key, group); + } + group.values.push(value); + } + } + + const stats: GpuMetricStats[] = []; + for (const group of groups.values()) { + const sorted = group.values.toSorted((a, b) => a - b); + const mean = sorted.reduce((acc, v) => acc + v, 0) / sorted.length; + const variance = sorted.reduce((acc, v) => acc + (v - mean) ** 2, 0) / sorted.length; + stats.push({ + gpuIndex: group.gpuIndex, + metric: group.metric, + count: sorted.length, + min: sorted[0]!, + max: sorted.at(-1)!, + mean, + median: percentile(sorted, 50), + p95: percentile(sorted, 95), + p99: percentile(sorted, 99), + stddev: Math.sqrt(variance), + }); + } + return stats.toSorted((a, b) => a.gpuIndex - b.gpuIndex || a.metric.localeCompare(b.metric)); +} + +export interface GpuMetricSeriesSummary { + sampleCount: number; + gpuCount: number; + startedAtMs: number; + endedAtMs: number; + /** Median gap between consecutive samples of one GPU, in seconds. */ + sampleIntervalS: number | null; +} + +/** Window and cadence facts stored on the series row. */ +export function summarizeGpuMetricSamples( + samples: readonly GpuMetricSample[], +): GpuMetricSeriesSummary | null { + if (samples.length === 0) return null; + let startedAtMs = Number.POSITIVE_INFINITY; + let endedAtMs = Number.NEGATIVE_INFINITY; + const byGpu = new Map(); + for (const sample of samples) { + if (sample.timestampMs < startedAtMs) startedAtMs = sample.timestampMs; + if (sample.timestampMs > endedAtMs) endedAtMs = sample.timestampMs; + const bucket = byGpu.get(sample.gpuIndex); + if (bucket) bucket.push(sample.timestampMs); + else byGpu.set(sample.gpuIndex, [sample.timestampMs]); + } + const gaps: number[] = []; + for (const timestamps of byGpu.values()) { + const sorted = timestamps.toSorted((a, b) => a - b); + for (let i = 1; i < sorted.length; i++) { + const gap = sorted[i]! - sorted[i - 1]!; + if (gap > 0) gaps.push(gap); + } + } + const sampleIntervalS = + gaps.length > 0 + ? percentile( + gaps.toSorted((a, b) => a - b), + 50, + ) / 1000 + : null; + return { + sampleCount: samples.length, + gpuCount: byGpu.size, + startedAtMs, + endedAtMs, + sampleIntervalS, + }; +} diff --git a/packages/db/src/etl/gpu-metrics-ingest.test.ts b/packages/db/src/etl/gpu-metrics-ingest.test.ts new file mode 100644 index 000000000..92d1506ba --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-ingest.test.ts @@ -0,0 +1,416 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { PGlite } from '@electric-sql/pglite'; +import type postgres from 'postgres'; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from 'vitest'; + +import { + ingestGpuMetricsArtifact, + prepareGpuMetricsArtifact, + refreshGpuMetricStats, +} from './gpu-metrics-ingest'; + +import { findOutdatedGpuMetricSeries } from '../lib/gpu-metrics-backfill'; +import { GPU_STATS_VERSION } from '../lib/gpu-metric-stats'; + +type Sql = postgres.Sql; +let db: PGlite; +let sql: Sql; +const roots: string[] = []; + +function queryClient(database: Pick) { + const client = async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await database.query(query, values); + return result.rows; + }; + return Object.assign(client, { + json: JSON.stringify, + array: (value: unknown) => value, + }); +} + +async function storedSeries(seriesId: number) { + return { + metadata: await sql`select * from gpu_metric_series where id = ${seriesId}`, + samples: await sql`select * from gpu_metric_samples where series_id = ${seriesId} + order by sampled_at, gpu_index`, + stats: await sql`select * from gpu_metric_gpu_stats where series_id = ${seriesId} + order by gpu_index, metric`, + links: await sql`select * from benchmark_result_gpu_metrics where series_id = ${seriesId} + order by benchmark_result_id`, + }; +} + +beforeAll(async () => { + db = await PGlite.create(); + for (const name of [ + '001_initial_schema.sql', + '016_gpu_metrics.sql', + '017_gpu_metric_stats_version.sql', + ]) { + await db.exec(fs.readFileSync(new URL(`../../migrations/${name}`, import.meta.url), 'utf8')); + } + sql = Object.assign(queryClient(db), { + begin: (fn: (tx: Sql) => Promise) => + db.transaction((tx) => fn(queryClient(tx) as unknown as Sql)), + }) as unknown as Sql; +}, 20_000); + +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, status, conclusion, created_at, date) + VALUES (1, 34557177019, 1, 'Run Sweep', 'completed', 'success', '2026-09-11', '2026-09-11'); + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, 4, 4, 4, 4); + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, isl, osl, conc, metrics) + VALUES (10, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, '{}'), + (11, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 64, '{}');`); +}); + +afterEach(() => { + for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); +}); + +afterAll(async () => { + await db?.close(); +}); + +const NVIDIA_CSV = [ + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]', + '2026/09/11 04:19:41.982, 0, 187.80 W, 33, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:41.986, 1, 190.96 W, 39, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:42.990, 0, 912.10 W, 61, 1965 MHz, 3996 MHz, 98 %, 74 %', + '2026/09/11 04:19:42.994, 1, 905.30 W, 62, 1965 MHz, 3996 MHz, 97 %, 73 %', + // Duplicate flush of the last sample, as emitted when the monitor stops. + '2026/09/11 04:19:42.994, 1, 905.30 W, 62, 1965 MHz, 3996 MHz, 97 %, 73 %', +].join('\n'); + +function writeArtifact(csv: string, contextZone = 'UTC') { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gpu-metrics-ingest-')); + roots.push(root); + const artifactName = 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0'; + const artifactDir = path.join(root, artifactName); + fs.mkdirSync(artifactDir); + fs.writeFileSync(path.join(artifactDir, 'gpu_metrics.csv'), csv); + fs.writeFileSync( + path.join(artifactDir, 'gpu_metrics_context.json'), + JSON.stringify({ timestamp_timezone: contextZone }), + ); + fs.writeFileSync( + path.join(artifactDir, 'gpu_metrics_identity.csv'), + 'index, uuid, name\n0, GPU-a, NVIDIA B200\n1, GPU-b, NVIDIA B200\n', + ); + return { artifactName, artifactDir }; +} + +// Multinode bundle: two hosts × two GPUs × two scrapes, host-local indices. +const MULTINODE_POWER_CSV = [ + 'schema_version,timestamp_unix,scrape_seq,hostname,gpu_index,gpu_uuid,power_w', + '1,1789194365.292,4,host-b,0,GPU-b0,401.5', + '1,1789194365.292,4,host-a,0,GPU-a0,186.656', + '1,1789194365.292,4,host-a,1,GPU-a1,190.1', + '1,1789194366.292,5,host-a,0,GPU-a0,700.25', + '1,1789194366.292,5,host-a,1,GPU-a1,702.0', + '1,1789194366.292,5,host-b,0,GPU-b0,650.0', + '1,1789194366.292,5,host-b,1,GPU-b1,655.0', + '1,1789194366.292,5,host-a,0,GPU-a0,999.0', +].join('\n'); +const MULTINODE_MANIFEST = { + schema_version: 1, + producer: 'srt-slurm.dcgm-power', + source_metric: 'DCGM_FI_DEV_POWER_USAGE', + sample_interval_seconds: 1, +}; + +function writePowerAuditArtifact() { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gpu-metrics-ingest-')); + roots.push(root); + const artifactName = 'power_audit_kimik3_conc8_fp4_dynamo-vllm_b200-slurm_0'; + const artifactDir = path.join(root, artifactName); + fs.mkdirSync(path.join(artifactDir, 'LOGS', 'power'), { recursive: true }); + fs.writeFileSync(path.join(artifactDir, 'LOGS', 'power', 'samples.csv'), MULTINODE_POWER_CSV); + fs.writeFileSync( + path.join(artifactDir, 'LOGS', 'power', 'manifest.json'), + JSON.stringify(MULTINODE_MANIFEST), + ); + fs.writeFileSync(path.join(artifactDir, 'agg_kimik3_conc8.json'), '{}'); + return { artifactName, artifactDir }; +} + +describe('prepareGpuMetricsArtifact', () => { + it('falls back to the multinode power bundle, one power-only series per host', () => { + const prepared = prepareGpuMetricsArtifact(writePowerAuditArtifact()); + expect(prepared.map((series) => series.fileName)).toEqual([ + 'LOGS/power/samples.csv#host-a', + 'LOGS/power/samples.csv#host-b', + ]); + const [hostA, hostB] = prepared; + expect(hostA!.vendor).toBe('nvidia'); + expect(hostA!.gpuCount).toBe(2); + expect(hostA!.samples).toHaveLength(4); + expect(hostA!.startedAtMs).toBe(1789194365292); + expect(hostA!.endedAtMs).toBe(1789194366292); + expect(hostA!.sampleIntervalS).toBe(1); + // One deployment-wide CSV: both host series share its hash. + expect(hostB!.csvSha256).toBe(hostA!.csvSha256); + // Only power was scraped; no zero-filled clock/temperature/utilization digests. + expect(new Set(hostA!.stats.map((s) => s.metric))).toEqual(new Set(['powerW'])); + expect(hostA!.stats.find((s) => s.gpuIndex === 1)?.max).toBe(702); + expect(hostA!.sidecars.context).toEqual(MULTINODE_MANIFEST); + expect(hostA!.sidecars.identity).toEqual([ + { hostname: 'host-a', gpu_index: 0, gpu_uuid: 'GPU-a0' }, + { hostname: 'host-a', gpu_index: 1, gpu_uuid: 'GPU-a1' }, + ]); + expect(hostA!.sidecars.energyStart).toBeNull(); + }); +}); + +describe('ingestGpuMetricsArtifact', () => { + it('refuses a partial artifact when a discovered host CSV has only a header', async () => { + const artifact = writeArtifact(NVIDIA_CSV); + const missingHostFile = path.join(artifact.artifactDir, 'host-b', 'gpu_metrics.csv'); + fs.mkdirSync(path.dirname(missingHostFile)); + fs.writeFileSync(missingHostFile, NVIDIA_CSV.split('\n')[0]!); + await expect( + ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10], + }), + ).rejects.toThrow('host-b/gpu_metrics.csv'); + const before = await sql`select count(*)::int as n from gpu_metric_series`; + expect(before).toEqual([{ n: 0 }]); + fs.writeFileSync(missingHostFile, NVIDIA_CSV); + const recovered = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10], + }); + expect(recovered.seriesIds).toHaveLength(2); + expect(recovered.samplesInserted).toBe(8); + const replay = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10], + }); + expect(replay.samplesInserted).toBe(0); + }); + + it('stores series, samples, digest, and point links; reruns are no-ops', async () => { + const artifact = writeArtifact(NVIDIA_CSV); + const first = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10, 10], + }); + expect(first.seriesIds).toHaveLength(1); + expect(first.samplesInserted).toBe(4); + expect(first.seriesSkipped).toBe(0); + + const [series] = await sql< + { + artifact_name: string; + config_key: string; + vendor: string; + sample_count: number; + gpu_count: number; + started_at: Date; + ended_at: Date; + sample_interval_s: number | null; + }[] + >`select artifact_name, config_key, vendor, sample_count, gpu_count, started_at, ended_at, + sample_interval_s from gpu_metric_series`; + expect(series!.artifact_name).toBe(artifact.artifactName); + expect(series!.config_key).toBe('dsr1_8k1k_fp4_sglang_conc32_b200-x_0'); + expect(series!.vendor).toBe('nvidia'); + expect(series!.sample_count).toBe(4); + expect(series!.gpu_count).toBe(2); + expect(new Date(series!.started_at).toISOString()).toBe('2026-09-11T04:19:41.982Z'); + expect(new Date(series!.ended_at).toISOString()).toBe('2026-09-11T04:19:42.994Z'); + expect(series!.sample_interval_s).toBeCloseTo(1.008, 3); + + const samples = await sql<{ gpu_index: number; power_w: number; sampled_at: Date }[]>` + select gpu_index, power_w, sampled_at from gpu_metric_samples order by sampled_at, gpu_index`; + expect(samples.map((s) => [s.gpu_index, s.power_w])).toEqual([ + [0, 187.8], + [1, 190.96], + [0, 912.1], + [1, 905.3], + ]); + + const stats = await sql< + { gpu_index: number; metric: string; sample_count: number; max_value: number }[] + >` + select gpu_index, metric, sample_count, max_value from gpu_metric_gpu_stats + where metric = 'power_w' order by gpu_index`; + expect(stats).toEqual([ + { gpu_index: 0, metric: 'power_w', sample_count: 2, max_value: 912.1 }, + { gpu_index: 1, metric: 'power_w', sample_count: 2, max_value: 905.3 }, + ]); + + const links = await sql<{ benchmark_result_id: number }[]>` + select benchmark_result_id from benchmark_result_gpu_metrics order by 1`; + expect(links.map((l) => Number(l.benchmark_result_id))).toEqual([10]); + + const rerun = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10, 11], + }); + expect(rerun.seriesIds).toEqual(first.seriesIds); + expect(rerun.samplesInserted).toBe(0); + expect(rerun.seriesSkipped).toBe(1); + const relinked = await sql<{ benchmark_result_id: number }[]>` + select benchmark_result_id from benchmark_result_gpu_metrics order by 1`; + expect(relinked.map((l) => Number(l.benchmark_result_id))).toEqual([10, 11]); + const [count] = await sql<{ n: number }[]>`select count(*)::int as n from gpu_metric_samples`; + expect(count!.n).toBe(4); + }); + + it('rolls back a replacement when point linking fails and retries without losing existing links', async () => { + const artifact = writeArtifact(NVIDIA_CSV); + const first = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10], + }); + const seriesId = first.seriesIds[0]!; + const before = await storedSeries(seriesId); + fs.writeFileSync( + path.join(artifact.artifactDir, 'gpu_metrics.csv'), + `${NVIDIA_CSV}\n2026/09/11 04:19:43.990, 0, 950.00 W, 65, 1965 MHz, 3996 MHz, 99 %, 75 %`, + ); + fs.writeFileSync( + path.join(artifact.artifactDir, 'gpu_metrics_context.json'), + JSON.stringify({ timestamp_timezone: '+02:00' }), + ); + + // Point links are written last, after replacement metadata, samples and stats. + await expect( + ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [11, 999], + }), + ).rejects.toMatchObject({ code: '23503' }); + expect(await storedSeries(seriesId)).toEqual(before); + + const recovered = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [11], + }); + expect(recovered).toEqual({ + metadataUpdatedBenchmarkResultIds: [], + seriesIds: [seriesId], + samplesInserted: 5, + seriesSkipped: 0, + }); + const after = await storedSeries(seriesId); + expect(after.metadata).toMatchObject([ + { sample_count: 5, sidecars: { context: { timestamp_timezone: '+02:00' } } }, + ]); + expect(after.metadata[0]!.csv_sha256).not.toBe(before.metadata[0]!.csv_sha256); + expect(after.samples).toHaveLength(5); + expect(new Date(after.samples[0]!.sampled_at).toISOString()).toBe('2026-09-11T02:19:41.982Z'); + expect( + after.stats.find((row) => row.gpu_index === 0 && row.metric === 'power_w'), + ).toMatchObject({ + sample_count: 3, + max_value: 950, + }); + expect(after.links.map((row) => Number(row.benchmark_result_id))).toEqual([10, 11]); + }); +}); + +describe('versioned full-record digests', () => { + it('upgrades unchanged input without rewriting samples or links, then becomes a no-op', async () => { + const input = { + workflowRunId: 1, + artifact: writeArtifact(NVIDIA_CSV), + benchmarkResultIds: [10, 11], + }; + const first = await ingestGpuMetricsArtifact(sql, input); + const id = first.seriesIds[0]!; + const original = await storedSeries(id); + await sql`update gpu_metric_series set stats_version = 0 where id = ${id}`; + await sql`update gpu_metric_gpu_stats set mean_value = -1 where series_id = ${id}`; + const updated = await ingestGpuMetricsArtifact(sql, input); + expect(updated).toMatchObject({ samplesInserted: 0, seriesSkipped: 0 }); + expect(await storedSeries(id)).toEqual(original); + expect(await ingestGpuMetricsArtifact(sql, input)).toMatchObject({ + samplesInserted: 0, + seriesSkipped: 1, + }); + }); + + it('backfills both retained hosts without artifacts; preserves null metrics, links and unrelated series', async () => { + const artifact = writePowerAuditArtifact(); + const first = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10, 11], + }); + const unrelated = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: writeArtifact(NVIDIA_CSV), + benchmarkResultIds: [10], + }); + const untouched = await storedSeries(unrelated.seriesIds[0]!); + const originals = await Promise.all(first.seriesIds.map(storedSeries)); + await sql`update gpu_metric_series set stats_version = 0 where id = any(${sql.array(first.seriesIds)}::bigint[])`; + await sql`update gpu_metric_gpu_stats set mean_value = -1 where series_id = any(${sql.array(first.seriesIds)}::bigint[])`; + fs.rmSync(artifact.artifactDir, { recursive: true }); + // No default retention cutoff: old runs remain repairable. + await sql`update workflow_runs set date = '2020-01-01' where id = 1`; + const targets = await findOutdatedGpuMetricSeries( + sql, + { run: null, attempt: null, artifact: null, fromRun: null, since: null }, + null, + ); + expect(targets.map((row) => Number(row.id))).toEqual(first.seriesIds); + expect( + await findOutdatedGpuMetricSeries( + sql, + { run: 34557177019, attempt: 2, artifact: null, fromRun: null, since: null }, + null, + ), + ).toEqual([]); + for (const [index, id] of first.seriesIds.entries()) { + expect(await refreshGpuMetricStats(sql, id)).toBe(true); + expect(await storedSeries(id)).toEqual(originals[index]); + expect(await refreshGpuMetricStats(sql, id)).toBe(false); + } + expect(await storedSeries(unrelated.seriesIds[0]!)).toEqual(untouched); + }); + + it('rolls back a failed digest replacement and permits retry; refuses incomplete samples', async () => { + const first = await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: writeArtifact(NVIDIA_CSV), + benchmarkResultIds: [10], + }); + const id = first.seriesIds[0]!; + await sql`update gpu_metric_series set stats_version = 0 where id = ${id}`; + await sql`update gpu_metric_gpu_stats set mean_value = -1 where series_id = ${id}`; + const before = await storedSeries(id); + await db.exec( + 'ALTER TABLE gpu_metric_gpu_stats ADD CONSTRAINT reject_stats CHECK (mean_value < 0)', + ); + await expect(refreshGpuMetricStats(sql, id)).rejects.toMatchObject({ code: '23514' }); + expect(await storedSeries(id)).toEqual(before); + await db.exec('ALTER TABLE gpu_metric_gpu_stats DROP CONSTRAINT reject_stats'); + expect(await refreshGpuMetricStats(sql, id)).toBe(true); + const recovered = await storedSeries(id); + expect(recovered.metadata[0]?.stats_version).toBe(GPU_STATS_VERSION); + await sql`update gpu_metric_series set stats_version = 0 where id = ${id}`; + await sql`delete from gpu_metric_samples where series_id = ${id} and gpu_index = 0`; + const incomplete = await storedSeries(id); + await expect(refreshGpuMetricStats(sql, id)).rejects.toThrow('missing samples'); + expect(await storedSeries(id)).toEqual(incomplete); + }); +}); diff --git a/packages/db/src/etl/gpu-metrics-ingest.ts b/packages/db/src/etl/gpu-metrics-ingest.ts new file mode 100644 index 000000000..a00fe9872 --- /dev/null +++ b/packages/db/src/etl/gpu-metrics-ingest.ts @@ -0,0 +1,451 @@ +/** + * Persist one gpu_metrics artifact: series metadata, full-resolution samples, + * the per-GPU statistics digest, and links to the benchmark points it covers. + * + * Idempotency: a series is identified by (workflow run, artifact name, CSV + * path). Re-ingesting an identical CSV, sidecars and sample count only refreshes + * the point links when the statistics version is current; a source change + * replaces samples and digest atomically. + */ + +import { createHash } from 'node:crypto'; +import fs from 'node:fs'; + +import type postgres from 'postgres'; + +import { + GPU_STATS_VERSION, + statMetricColumn, + computeStoredGpuMetricStats, + type StoredGpuMetricSample, +} from '../lib/gpu-metric-stats.js'; + +export { statMetricColumn } from '../lib/gpu-metric-stats.js'; + +import type { Sql } from './db-utils.js'; + +/** Either a pooled client or the transaction handle passed to `sql.begin` callbacks. */ +type TxLike = Sql | postgres.TransactionSql; +import { + computeGpuMetricStats, + parseGpuMetricsCsv, + summarizeGpuMetricSamples, + type GpuMetricSample, + type GpuMetricStats, + type GpuMetricsVendor, +} from './gpu-metrics-csv.js'; +import { + contextUtcOffsetMinutes, + gpuMetricsArtifactSuffix, + listGpuMetricsCsvFiles, + listMultinodePowerSampleFiles, + readGpuMetricsSidecars, + readMultinodePowerManifest, + readPowerAuditValidations, + type GpuMetricsArtifact, + type GpuMetricsSidecars, +} from './gpu-metrics-artifacts.js'; +import { multinodePowerVendor, parseMultinodePowerSamples } from './multinode-power-samples.js'; +import { recoveredPowerAudit } from './power-audit-validations.js'; + +/** Samples are streamed to Postgres in unnest batches of this many rows. */ +const SAMPLE_BATCH_SIZE = 5000; + +function uniqueSamples(samples: readonly GpuMetricSample[]): GpuMetricSample[] { + const seen = new Set(); + return samples.filter((sample) => { + const key = `${sample.gpuIndex}:${sample.timestampMs}`; + if (seen.has(key)) return false; + // Keep the first row, matching INSERT ... ON CONFLICT DO NOTHING. + seen.add(key); + return true; + }); +} + +export interface PreparedGpuMetricSeries { + fileName: string; + vendor: GpuMetricsVendor; + csvSha256: string; + samples: GpuMetricSample[]; + stats: GpuMetricStats[]; + sampleIntervalS: number | null; + gpuCount: number; + startedAtMs: number; + endedAtMs: number; + sidecars: GpuMetricsSidecars; +} + +export interface GpuMetricsIngestResult { + seriesIds: number[]; + samplesInserted: number; + seriesSkipped: number; + metadataUpdatedBenchmarkResultIds: number[]; +} + +/** + * One series per host from the multinode power bundle. The deployment-wide + * CSV is hashed once, so every host series of one upload shares its source sha. + * The `#` fragment keeps the + * (run, artifact, file) identity unique per host. + */ +function prepareMultinodePowerSeries(artifact: GpuMetricsArtifact): PreparedGpuMetricSeries[] { + const prepared: PreparedGpuMetricSeries[] = []; + const validations = readPowerAuditValidations(artifact.artifactDir, artifact.artifactName); + for (const file of listMultinodePowerSampleFiles(artifact.artifactDir)) { + const csvText = fs.readFileSync(file.path, 'utf8'); + const hosts = parseMultinodePowerSamples(csvText); + if (!hosts) continue; + const manifest = readMultinodePowerManifest(file.path); + const vendor = multinodePowerVendor(manifest); + const csvSha256 = createHash('sha256').update(csvText).digest('hex'); + for (const host of hosts) { + const samples = uniqueSamples(host.samples); + const summary = summarizeGpuMetricSamples(samples); + if (!summary) continue; + prepared.push({ + fileName: `${file.fileName}#${host.hostname}`, + vendor, + csvSha256, + samples, + stats: computeGpuMetricStats(samples), + sampleIntervalS: summary.sampleIntervalS, + gpuCount: summary.gpuCount, + startedAtMs: summary.startedAtMs, + endedAtMs: summary.endedAtMs, + sidecars: { + context: manifest, + validations, + identity: Object.entries(host.gpuUuids).map(([index, uuid]) => ({ + hostname: host.hostname, + gpu_index: Number(index), + gpu_uuid: uuid, + })), + energyStart: null, + energyEnd: null, + }, + }); + } + } + return prepared; +} + +/** + * Parse every CSV in an extracted artifact; refuse a partly readable CSV set. + * Without usable nvidia-smi/amd-smi CSVs, fall back to the multinode power bundle. + */ +export function prepareGpuMetricsArtifact(artifact: GpuMetricsArtifact): PreparedGpuMetricSeries[] { + const prepared: PreparedGpuMetricSeries[] = []; + const unreadable: string[] = []; + for (const file of listGpuMetricsCsvFiles(artifact.artifactDir)) { + const csvText = fs.readFileSync(file.path, 'utf8'); + const sidecars = readGpuMetricsSidecars(file.path); + if (artifact.artifactName.startsWith('power_audit_')) { + sidecars.validations = readPowerAuditValidations(artifact.artifactDir, artifact.artifactName); + const bundleFile = listMultinodePowerSampleFiles(artifact.artifactDir)[0]; + sidecars.powerManifest = bundleFile ? readMultinodePowerManifest(bundleFile.path) : null; + } + const parsed = parseGpuMetricsCsv(csvText, { + nvidiaUtcOffsetMinutes: contextUtcOffsetMinutes(sidecars.context), + }); + if (!parsed) { + unreadable.push(file.fileName); + continue; + } + const samples = uniqueSamples(parsed.samples); + const summary = summarizeGpuMetricSamples(samples); + if (!summary) { + unreadable.push(file.fileName); + continue; + } + prepared.push({ + fileName: file.fileName, + vendor: parsed.vendor, + csvSha256: createHash('sha256').update(csvText).digest('hex'), + samples, + stats: computeGpuMetricStats(samples), + sampleIntervalS: summary.sampleIntervalS, + gpuCount: summary.gpuCount, + startedAtMs: summary.startedAtMs, + endedAtMs: summary.endedAtMs, + sidecars, + }); + } + if (prepared.length > 0 && unreadable.length > 0) { + throw new Error( + `Incomplete telemetry artifact ${artifact.artifactName}: unreadable CSVs ${unreadable.join(', ')}`, + ); + } + const series = prepared.length > 0 ? prepared : prepareMultinodePowerSeries(artifact); + const seriesInventory = series.map((entry) => ({ + fileName: entry.fileName, + sampleCount: entry.samples.length, + })); + for (const entry of series) entry.sidecars.seriesInventory = seriesInventory; + return series; +} + +async function insertSampleBatch( + tx: TxLike, + seriesId: number, + batch: readonly GpuMetricSample[], +): Promise { + const inserted = await tx<{ n: number }[]>` + with ins as ( + insert into gpu_metric_samples ( + series_id, gpu_index, sampled_at, + power_w, temperature_c, sm_clock_mhz, mem_clock_mhz, gpu_util_pct, mem_util_pct, + edge_temp_c, mem_temp_c, gfx_voltage_mv, soc_voltage_mv, mem_voltage_mv, + fclk_mhz, socclk_mhz, mm_activity_pct + ) + select + ${seriesId}, + unnest(${tx.array(batch.map((s) => s.gpuIndex))}::smallint[]), + to_timestamp(unnest(${tx.array(batch.map((s) => s.timestampMs / 1000))}::double precision[])), + unnest(${tx.array(batch.map((s) => s.powerW))}::real[]), + unnest(${tx.array(batch.map((s) => s.temperatureC))}::real[]), + unnest(${tx.array(batch.map((s) => s.smClockMhz))}::real[]), + unnest(${tx.array(batch.map((s) => s.memClockMhz))}::real[]), + unnest(${tx.array(batch.map((s) => s.gpuUtilPct))}::real[]), + unnest(${tx.array(batch.map((s) => s.memUtilPct))}::real[]), + unnest(${tx.array(batch.map((s) => s.edgeTempC))}::real[]), + unnest(${tx.array(batch.map((s) => s.memTempC))}::real[]), + unnest(${tx.array(batch.map((s) => s.gfxVoltageMv))}::real[]), + unnest(${tx.array(batch.map((s) => s.socVoltageMv))}::real[]), + unnest(${tx.array(batch.map((s) => s.memVoltageMv))}::real[]), + unnest(${tx.array(batch.map((s) => s.fclkMhz))}::real[]), + unnest(${tx.array(batch.map((s) => s.socclkMhz))}::real[]), + unnest(${tx.array(batch.map((s) => s.mmActivityPct))}::real[]) + -- nvidia-smi occasionally repeats the final sample when the monitor is + -- stopped and flushed; the primary key makes that a no-op. + on conflict (series_id, gpu_index, sampled_at) do nothing + returning 1 + ) + select count(*)::int as n from ins + `; + return Number(inserted[0]?.n ?? 0); +} + +async function insertStats( + tx: TxLike, + seriesId: number, + stats: readonly GpuMetricStats[], +): Promise { + if (stats.length === 0) return; + await tx` + insert into gpu_metric_gpu_stats ( + series_id, gpu_index, metric, sample_count, + min_value, max_value, mean_value, median_value, p95_value, p99_value, stddev_value + ) + select + ${seriesId}, + unnest(${tx.array(stats.map((s) => s.gpuIndex))}::smallint[]), + unnest(${tx.array(stats.map((s) => statMetricColumn(s.metric)))}::text[]), + unnest(${tx.array(stats.map((s) => s.count))}::int[]), + unnest(${tx.array(stats.map((s) => s.min))}::real[]), + unnest(${tx.array(stats.map((s) => s.max))}::real[]), + unnest(${tx.array(stats.map((s) => s.mean))}::real[]), + unnest(${tx.array(stats.map((s) => s.median))}::real[]), + unnest(${tx.array(stats.map((s) => s.p95))}::real[]), + unnest(${tx.array(stats.map((s) => s.p99))}::real[]), + unnest(${tx.array(stats.map((s) => s.stddev))}::real[]) + `; +} + +/** + * Upsert one prepared series and link it to `benchmarkResultIds`. Returns the + * series id and how many sample rows were written (0 when the CSV, sidecars, + * and unique sample count are unchanged). + */ +export function upsertGpuMetricSeries( + sql: Sql, + input: { + workflowRunId: number; + artifactName: string; + series: PreparedGpuMetricSeries; + benchmarkResultIds: readonly number[]; + }, +): Promise<{ + seriesId: number; + samplesInserted: number; + replaced: boolean; + statsUpdated: boolean; +}> { + const { workflowRunId, artifactName, series, benchmarkResultIds } = input; + const configKey = gpuMetricsArtifactSuffix(artifactName) ?? artifactName; + const sidecarsJson = JSON.stringify(series.sidecars); + + return sql.begin(async (tx) => { + const existing = await tx< + { + id: number; + csv_sha256: string; + sample_count: number; + stats_version: number; + sidecars_match: boolean; + }[] + >` + select id, csv_sha256, sample_count, stats_version, sidecars = ${sidecarsJson}::jsonb as sidecars_match + from gpu_metric_series + where workflow_run_id = ${workflowRunId} + and artifact_name = ${artifactName} + and file_name = ${series.fileName} + for update + `; + + let seriesId: number; + let needsSamples = true; + let replaced = false; + if (existing.length > 0) { + seriesId = Number(existing[0]!.id); + // Count also detects legacy digests computed before duplicate removal. + if ( + existing[0]!.csv_sha256 === series.csvSha256 && + existing[0]!.sidecars_match && + existing[0]!.sample_count === series.samples.length + ) { + needsSamples = false; + } else { + replaced = true; + await tx`delete from gpu_metric_samples where series_id = ${seriesId}`; + await tx`delete from gpu_metric_gpu_stats where series_id = ${seriesId}`; + await tx` + update gpu_metric_series set + vendor = ${series.vendor}, + csv_sha256 = ${series.csvSha256}, + sample_interval_s = ${series.sampleIntervalS}, + sample_count = ${series.samples.length}, + gpu_count = ${series.gpuCount}, + started_at = to_timestamp(${series.startedAtMs / 1000}::double precision), + ended_at = to_timestamp(${series.endedAtMs / 1000}::double precision), + sidecars = ${sidecarsJson}::jsonb, + ingested_at = now() + where id = ${seriesId} + `; + } + } else { + const [row] = await tx<{ id: number }[]>` + insert into gpu_metric_series ( + workflow_run_id, artifact_name, config_key, file_name, vendor, csv_sha256, + sample_interval_s, sample_count, gpu_count, started_at, ended_at, sidecars + ) values ( + ${workflowRunId}, ${artifactName}, ${configKey}, ${series.fileName}, + ${series.vendor}, ${series.csvSha256}, ${series.sampleIntervalS}, + ${series.samples.length}, ${series.gpuCount}, + to_timestamp(${series.startedAtMs / 1000}::double precision), + to_timestamp(${series.endedAtMs / 1000}::double precision), + ${sidecarsJson}::jsonb + ) + returning id + `; + seriesId = Number(row!.id); + } + + let samplesInserted = 0; + if (needsSamples) { + for (let offset = 0; offset < series.samples.length; offset += SAMPLE_BATCH_SIZE) { + samplesInserted += await insertSampleBatch( + tx, + seriesId, + series.samples.slice(offset, offset + SAMPLE_BATCH_SIZE), + ); + } + } + const statsUpdated = needsSamples || existing[0]?.stats_version !== GPU_STATS_VERSION; + if (statsUpdated) { + if (!needsSamples) await tx`delete from gpu_metric_gpu_stats where series_id = ${seriesId}`; + await insertStats(tx, seriesId, series.stats); + await tx`update gpu_metric_series set stats_version = ${GPU_STATS_VERSION} where id = ${seriesId}`; + } + + if (benchmarkResultIds.length > 0) { + await tx` + insert into benchmark_result_gpu_metrics (benchmark_result_id, series_id) + select unnest(${tx.array([...new Set(benchmarkResultIds)])}::bigint[]), ${seriesId} + on conflict do nothing + `; + } + + return { seriesId, samplesInserted, replaced, statsUpdated }; + }); +} + +export function refreshGpuMetricStats(sql: Sql, seriesId: number): Promise { + return sql.begin(async (tx) => { + const [series] = await tx<{ sample_count: number; stats_version: number }[]>` + select sample_count, stats_version from gpu_metric_series where id = ${seriesId} for update + `; + if (!series) throw new Error(`Unknown telemetry series ${seriesId}`); + if (series.stats_version === GPU_STATS_VERSION) return false; + const samples = await tx` + select * from gpu_metric_samples where series_id = ${seriesId} order by sampled_at, gpu_index + `; + if (samples.length !== Number(series.sample_count)) { + throw new Error( + `Telemetry series ${seriesId} has missing samples; re-ingest its source artifact`, + ); + } + const stats = computeStoredGpuMetricStats(samples); + await tx`delete from gpu_metric_gpu_stats where series_id = ${seriesId}`; + await insertStats(tx, seriesId, stats); + await tx`update gpu_metric_series set stats_version = ${GPU_STATS_VERSION} where id = ${seriesId}`; + return true; + }); +} + +/** Read, digest, and persist every CSV of one artifact for one set of points. */ +export async function ingestGpuMetricsArtifact( + sql: Sql, + input: { + workflowRunId: number; + artifact: GpuMetricsArtifact; + benchmarkResultIds: readonly number[]; + }, +): Promise { + const prepared = prepareGpuMetricsArtifact(input.artifact); + const result: GpuMetricsIngestResult = { + seriesIds: [], + samplesInserted: 0, + seriesSkipped: 0, + metadataUpdatedBenchmarkResultIds: [], + }; + for (const series of prepared) { + const upserted = await upsertGpuMetricSeries(sql, { + workflowRunId: input.workflowRunId, + artifactName: input.artifact.artifactName, + series, + benchmarkResultIds: input.benchmarkResultIds, + }); + result.seriesIds.push(upserted.seriesId); + result.samplesInserted += upserted.samplesInserted; + if (upserted.samplesInserted === 0 && !upserted.replaced && !upserted.statsUpdated) + result.seriesSkipped++; + } + // The caller has resolved this artifact's exact benchmark identities. Recover + // only a unique AgentX point within that explicit set, after every host is + // stored/linked. This also repairs metadata when all samples were a no-op. + const validations = prepared[0]?.sidecars.validations ?? {}; + const audits = Object.entries(validations) + .map(([source, validation]) => ({ + audit: recoveredPowerAudit(source, validation), + conc: (validation.selected_window as Record | undefined)?.concurrency, + })) + .filter((entry) => entry.audit !== null); + for (const { audit, conc } of audits) { + if (typeof conc !== 'number' || !Number.isSafeInteger(conc) || conc <= 0) continue; + if (audits.filter((entry) => entry.conc === conc).length !== 1) continue; + const updated = await sql<{ id: number }[]>` + with candidates as ( + select id from benchmark_results + where workflow_run_id = ${input.workflowRunId} + and id = any(${sql.array([...new Set(input.benchmarkResultIds)])}::bigint[]) + and benchmark_type = 'agentic_traces' and conc = ${conc} + ) + update benchmark_results set power_audit = ${sql.json(audit)}::jsonb + where id in (select id from candidates) + and (select count(*) from candidates) = 1 and power_audit is null + returning id + `; + result.metadataUpdatedBenchmarkResultIds.push(...updated.map((row) => Number(row.id))); + } + return result; +} diff --git a/packages/db/src/etl/ingest-summary.ts b/packages/db/src/etl/ingest-summary.ts index 16806cc47..684d7a1f5 100644 --- a/packages/db/src/etl/ingest-summary.ts +++ b/packages/db/src/etl/ingest-summary.ts @@ -38,6 +38,7 @@ export function printIngestSummaryFooter( ['unmapped hw', skips.unmappedHw], ['bad/empty zip', skips.badZip], ['DB errors', skips.dbError], + ['telemetry errors (non-fatal)', skips.telemetryError], ); const nonzeroSkipLines = skipLines.filter(([, count]) => count > 0); diff --git a/packages/db/src/etl/multinode-power-samples.ts b/packages/db/src/etl/multinode-power-samples.ts new file mode 100644 index 000000000..9c711c4de --- /dev/null +++ b/packages/db/src/etl/multinode-power-samples.ts @@ -0,0 +1,114 @@ +/** + * Parser for the multinode power producer's `LOGS/power/samples.csv`. + * + * `benchmark-multinode-tmpl.yml` uploads no `gpu_metrics_` artifact; its + * telemetry travels inside `power_audit_` as one CSV for the whole + * deployment, written by `srt-slurm.dcgm-power` (DCGM scraped on every host + * once per second): + * + * schema_version,timestamp_unix,scrape_seq,hostname,gpu_index,gpu_uuid,power_w + * + * Only power is scraped, so every other `GpuMetricSample` field stays null and + * the digest carries `powerW` alone. Rows are regrouped per host so each host + * becomes its own series, the shape the reader already uses for multinode + * staging ("one CSV per node"). Pure module: no I/O. + */ + +import { splitCsvLine, type GpuMetricSample, type GpuMetricsVendor } from './gpu-metrics-csv.js'; + +export interface MultinodePowerHost { + hostname: string; + /** Host-local GPU indices, as DCGM reports them. */ + samples: GpuMetricSample[]; + /** gpu_index → gpu_uuid for the identity sidecar. */ + gpuUuids: Record; +} + +const REQUIRED_COLUMNS = ['timestamp_unix', 'hostname', 'gpu_index', 'power_w'] as const; + +/** `samples.csv` under `LOGS/` is the producer's only time-series file. */ +export function isMultinodePowerSamplesPath(relativePath: string): boolean { + const posix = relativePath.split('\\').join('/'); + return /^LOGS\/(?:[^/]+\/)*samples\.csv$/u.test(posix); +} + +function powerOnlySample(timestampMs: number, gpuIndex: number, powerW: number): GpuMetricSample { + return { + timestampMs, + gpuIndex, + powerW, + temperatureC: null, + smClockMhz: null, + memClockMhz: null, + gpuUtilPct: null, + memUtilPct: null, + edgeTempC: null, + memTempC: null, + gfxVoltageMv: null, + socVoltageMv: null, + memVoltageMv: null, + fclkMhz: null, + socclkMhz: null, + mmActivityPct: null, + }; +} + +/** + * Group the deployment-wide CSV by host. Returns null when the header is not + * the multinode power schema; malformed rows are skipped. Hosts sort by name + * so series order is stable across re-ingests. + */ +export function parseMultinodePowerSamples(csvText: string): MultinodePowerHost[] | null { + const lines = csvText + .split('\n') + .map((line) => line.replace(/\r$/u, '')) + .filter((line) => line.trim().length > 0); + if (lines.length <= 1) return null; + const header = splitCsvLine(lines[0]!).map((cell) => cell.trim().toLowerCase()); + const column = new Map(header.map((name, index) => [name, index] as const)); + if (REQUIRED_COLUMNS.some((name) => !column.has(name))) return null; + const at = (cells: string[], name: string): string | undefined => cells[column.get(name)!]; + + const hosts = new Map(); + for (const line of lines.slice(1)) { + const cells = splitCsvLine(line).map((cell) => cell.trim()); + const hostname = at(cells, 'hostname'); + const seconds = Number.parseFloat(at(cells, 'timestamp_unix') ?? ''); + const gpuIndex = Number.parseInt(at(cells, 'gpu_index') ?? '', 10); + const powerW = Number.parseFloat(at(cells, 'power_w') ?? ''); + if ( + !hostname || + !Number.isFinite(seconds) || + !Number.isSafeInteger(gpuIndex) || + gpuIndex < 0 || + !Number.isFinite(powerW) + ) { + continue; + } + let host = hosts.get(hostname); + if (!host) { + host = { hostname, samples: [], gpuUuids: {} }; + hosts.set(hostname, host); + } + host.samples.push(powerOnlySample(Math.round(seconds * 1000), gpuIndex, powerW)); + const uuid = at(cells, 'gpu_uuid'); + if (uuid && !(gpuIndex in host.gpuUuids)) host.gpuUuids[gpuIndex] = uuid; + } + if (hosts.size === 0) return null; + return [...hosts.values()].toSorted((a, b) => a.hostname.localeCompare(b.hostname)); +} + +/** + * The producer manifest names its source metric; DCGM means NVIDIA. Anything + * naming AMD tooling maps to 'amd', and an absent manifest defaults to NVIDIA + * because only the DCGM producer writes this layout today. + */ +export function multinodePowerVendor(manifest: Record | null): GpuMetricsVendor { + const source = [manifest?.source_metric, manifest?.producer] + .filter((value): value is string => typeof value === 'string') + .join(' ') + .toLowerCase(); + // `amd-smi`, `rocm_smi`: the tool name may continue with `-` or `_`. + if (/\b(?:amd|rocm)/u.test(source)) return 'amd'; + return 'nvidia'; +} diff --git a/packages/db/src/etl/power-audit-recovery.test.ts b/packages/db/src/etl/power-audit-recovery.test.ts new file mode 100644 index 000000000..f90d1a0ae --- /dev/null +++ b/packages/db/src/etl/power-audit-recovery.test.ts @@ -0,0 +1,61 @@ +import { describe, expect, it } from 'vitest'; + +import type { BenchmarkPersistenceInput } from './benchmark-ingest.js'; +import { createBenchmarkPowerAuditRecovery } from './power-audit-recovery.js'; + +const resultFile = 'kimik3_recipe-a_conc48.json'; +const source = `power_validation_${resultFile}`; +const validation = { + power_valid: true, + validation_path: 'LOGS/agentic/conc_48/power_validation.json', + result_file: resultFile, + selected_window: { + concurrency: 48, + start_time_unix: 1000, + end_time_unix: 1100, + result_path: 'agentic/conc_48/agentic_power_concurrency_48.json', + window_file: 'windows/agentic_power_concurrency_48.json', + }, +}; +const evidence = { resultFile, validations: { [source]: validation } }; +const expectedAudit = { source, window_start_unix: 1000, window_end_unix: 1100 }; + +function point(overrides: Partial = {}): BenchmarkPersistenceInput { + return { + configId: 1, + benchmarkType: 'agentic_traces', + isl: null, + osl: null, + conc: 48, + offloadMode: 'on', + image: 'image', + recipeFingerprint: 'recipe-a', + metrics: { power_valid: 1, avg_power_w: 234.5, mean_ttft: 0.2 }, + workers: [{ role: 'agg', worker_idx: 0, num_gpus: 8, avg_power_w: 234.5 }], + ...overrides, + }; +} + +describe('CI benchmark audit recovery', () => { + it('retains exact provenance with aggregate after sibling and preserves benchmark values', () => { + const recover = createBenchmarkPowerAuditRecovery(); + const original = point(); + let latest = original; + for (const isPaired of [true, false]) { + latest = recover(original, isPaired ? evidence : undefined); + expect(latest.metrics).toBe(original.metrics); + expect(latest.workers).toBe(original.workers); + expect({ ...latest, powerAudit: undefined }).toEqual({ ...original, powerAudit: undefined }); + } + expect(latest.powerAudit).toEqual(expectedAudit); + expect(original.powerAudit).toBeUndefined(); + }); + + it('does not reuse the same run/concurrency audit for a point with another offload mode', () => { + const recover = createBenchmarkPowerAuditRecovery(); + recover(point(), evidence); + const other = point({ offloadMode: 'off' }); + expect(recover(other)).toBe(other); + expect(other.powerAudit).toBeUndefined(); + }); +}); diff --git a/packages/db/src/etl/power-audit-recovery.ts b/packages/db/src/etl/power-audit-recovery.ts new file mode 100644 index 000000000..808be3ef7 --- /dev/null +++ b/packages/db/src/etl/power-audit-recovery.ts @@ -0,0 +1,42 @@ +import { benchmarkPointIngestKey, type BenchmarkPersistenceInput } from './benchmark-ingest.js'; +import type { PowerAudit } from './benchmark-mapper.js'; +import { recoveredPowerAudit } from './power-audit-validations.js'; + +export interface BenchmarkPowerAuditEvidence { + resultFile: string; + validations: Record>; +} + +/** One run's exact sibling evidence also applies to its later aggregate copies. */ +export function createBenchmarkPowerAuditRecovery() { + const recovered = new Map(); + return ( + row: T, + evidence?: BenchmarkPowerAuditEvidence, + ): T => { + if (row.benchmarkType !== 'agentic_traces' || row.powerAudit !== undefined) return row; + const key = benchmarkPointIngestKey(row); + if (evidence) { + const matches = Object.entries(evidence.validations).filter(([, validation]) => { + const window = validation.selected_window; + if ( + validation.result_file !== evidence.resultFile || + !window || + typeof window !== 'object' || + Array.isArray(window) || + !('concurrency' in window) || + window.concurrency !== row.conc + ) + return false; + return true; + }); + // Ambiguous/missing evidence must not establish new provenance. + if (matches.length !== 1) return row; + const audit = recoveredPowerAudit(...matches[0]); + if (!audit) return row; + recovered.set(key, audit); + } + const audit = recovered.get(key); + return audit ? { ...row, powerAudit: audit } : row; + }; +} diff --git a/packages/db/src/etl/power-audit-validations.test.ts b/packages/db/src/etl/power-audit-validations.test.ts new file mode 100644 index 000000000..426073200 --- /dev/null +++ b/packages/db/src/etl/power-audit-validations.test.ts @@ -0,0 +1,67 @@ +import { describe, expect, it } from 'vitest'; + +import { normalizePowerAuditValidations } from './power-audit-validations.js'; + +const ARTIFACT = 'power_audit_kimik3_agentx_fp4'; +const RESULT = 'kimik3_agentx_fp4_conc48.json'; +const SOURCE = 'power_validation_kimik3_agentx_fp4_conc48.json'; +const VALIDATION = 'LOGS/agentic/conc_48/power_validation.json'; +const WINDOW = 'LOGS/power/windows/agentic_power_concurrency_48.json'; +const RESULT_PATH = 'agentic/conc_48/agentic_power_concurrency_48.json'; +const START = 1789940129.369988; +const END = 1789943769.371839; + +function fixture() { + const selected: Record = { + window_file: 'windows/agentic_power_concurrency_48.json', + result_path: RESULT_PATH, + benchmark_type: 'custom', + concurrency: 48, + start_time_unix: START, + end_time_unix: END, + duration: END - START, + }; + const validation: Record = { + power_valid: true, + selected_window: selected, + per_gpu_role: { 'host-a/GPU-a': 'prefill', 'host-b/GPU-b': 'decode' }, + validation_errors: [], + }; + const result: Record = { conc: 48, power_valid: 1 }; + const window: Record = { + schema_version: 1, + concurrency: 48, + result_path: RESULT_PATH, + benchmark_start_time_unix: START, + benchmark_end_time_unix: END, + status: 'completed', + }; + const files = () => + new Map([ + [VALIDATION, JSON.stringify(validation)], + [RESULT, JSON.stringify(result)], + [WINDOW, JSON.stringify(window)], + ]); + return { selected, validation, result, window, files }; +} + +describe('normalizePowerAuditValidations', () => { + it('aliases an exactly matched nested document without changing its window, roles or verdict', () => { + const input = fixture(); + const files = input.files(); + const retained = [...files]; + expect(normalizePowerAuditValidations(ARTIFACT, files)).toEqual( + new Map([ + [SOURCE, { ...input.validation, validation_path: VALIDATION, result_file: RESULT }], + ]), + ); + expect([...files]).toEqual(retained); + expect(input.validation).not.toHaveProperty('validation_path'); + }); + + it('rejects mismatched window.benchmark_start_time_unix', () => { + const input = fixture(); + input.window.benchmark_start_time_unix = START + 1; + expect(normalizePowerAuditValidations(ARTIFACT, input.files()).size).toBe(0); + }); +}); diff --git a/packages/db/src/etl/power-audit-validations.ts b/packages/db/src/etl/power-audit-validations.ts new file mode 100644 index 000000000..179c2abf1 --- /dev/null +++ b/packages/db/src/etl/power-audit-validations.ts @@ -0,0 +1,125 @@ +/** Pure normalization of legacy and retained AgentX power-audit documents. */ +const LEGACY_VALIDATION = /^power_validation_[^/]+\.json$/u; +const AGENTX_VALIDATION = /^LOGS\/agentic\/conc_(?[1-9]\d*)\/power_validation\.json$/u; +const AGENTX_WINDOW = /^LOGS\/power\/windows\/agentic_power_concurrency_[1-9]\d*\.json$/u; +const RESULT_FILE = /^(?[^/]+)_conc(?[1-9]\d*)\.json$/u; + +function isRecord(value: unknown): value is Record { + return value !== null && typeof value === 'object' && !Array.isArray(value); +} + +function parseObject(text: string | undefined): Record | null { + if (text === undefined) return null; + try { + const value: unknown = JSON.parse(text); + return isRecord(value) ? value : null; + } catch { + return null; + } +} + +function hasWindowTimes(value: unknown): value is Record & { + start_time_unix: number; + end_time_unix: number; +} { + return ( + isRecord(value) && + typeof value.start_time_unix === 'number' && + Number.isFinite(value.start_time_unix) && + value.start_time_unix > 0 && + typeof value.end_time_unix === 'number' && + Number.isFinite(value.end_time_unix) && + value.end_time_unix > value.start_time_unix + ); +} + +/** Only documents needed to identify a validation and its recorded result/window. */ +export function isPowerAuditValidationEntry(name: string): boolean { + return ( + LEGACY_VALIDATION.test(name) || + AGENTX_VALIDATION.test(name) || + AGENTX_WINDOW.test(name) || + RESULT_FILE.test(name) + ); +} + +/** + * Legacy top-level documents keep their names and contents. A nested AgentX + * document gains a canonical alias only when its result and retained window + * agree; path annotations retain the original document's provenance. + */ +export function normalizePowerAuditValidations( + artifactName: string, + files: ReadonlyMap, +): Map> { + const validations = new Map>(); + for (const [name, text] of files) { + if (!LEGACY_VALIDATION.test(name)) continue; + const validation = parseObject(text); + if (validation) validations.set(name, validation); + } + const artifactStem = /^power_audit_(?[^/]+)$/u.exec(artifactName)?.groups?.stem; + if (!artifactStem) return validations; + + for (const [validationPath, text] of files) { + const match = AGENTX_VALIDATION.exec(validationPath); + if (!match?.groups) continue; + const concurrency = Number(match.groups.concurrency); + if (!Number.isSafeInteger(concurrency)) continue; + const resultFile = `${artifactStem}_conc${concurrency}.json`; + const source = `power_validation_${resultFile.slice(0, -5)}.json`; + if (validations.has(source)) continue; + const validation = parseObject(text); + const result = parseObject(files.get(resultFile)); + const selected = validation?.selected_window; + const resultPath = `agentic/conc_${concurrency}/agentic_power_concurrency_${concurrency}.json`; + const windowFile = `windows/agentic_power_concurrency_${concurrency}.json`; + const window = parseObject(files.get(`LOGS/power/${windowFile}`)); + if ( + !validation || + result?.conc !== concurrency || + !hasWindowTimes(selected) || + selected.concurrency !== concurrency || + selected.result_path !== resultPath || + selected.window_file !== windowFile || + window?.concurrency !== concurrency || + window.result_path !== resultPath || + window.benchmark_start_time_unix !== selected.start_time_unix || + window.benchmark_end_time_unix !== selected.end_time_unix + ) + continue; + validations.set(source, { + ...validation, + validation_path: validationPath, + result_file: resultFile, + }); + } + return validations; +} + +/** Recover source/window metadata only from a normalized nested validation. */ +export function recoveredPowerAudit( + source: string, + validation: Record, +): { source: string; window_start_unix: number; window_end_unix: number } | null { + const path = validation.validation_path; + const result = validation.result_file; + if (typeof path !== 'string' || typeof result !== 'string') return null; + const nested = AGENTX_VALIDATION.exec(path)?.groups; + const file = RESULT_FILE.exec(result)?.groups; + const selected = validation.selected_window; + if ( + !nested || + !file || + nested.concurrency !== file.concurrency || + source !== `power_validation_${result.slice(0, -5)}.json` || + !hasWindowTimes(selected) || + selected.concurrency !== Number(nested.concurrency) + ) + return null; + return { + source, + window_start_unix: selected.start_time_unix, + window_end_unix: selected.end_time_unix, + }; +} diff --git a/packages/db/src/etl/power-publication.test.ts b/packages/db/src/etl/power-publication.test.ts index f223a07a4..031802d44 100644 --- a/packages/db/src/etl/power-publication.test.ts +++ b/packages/db/src/etl/power-publication.test.ts @@ -37,7 +37,7 @@ function expected(overrides = {}) { 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/123/attempts/2', { path: 'bmk_qwen3.5/results.json', sha256: 'abc' }, ); - if (!point) throw new Error('Fixture must be 8K/1K'); + if (!point) throw new Error('Fixture must belong to a supported PowerX workload'); return point; } function actual(point = expected()): PublishedPowerRow { @@ -51,6 +51,12 @@ function actual(point = expected()): PublishedPowerRow { } describe('PowerX publication', () => { + it('keeps required 1K/1K measurements in the publication receipt', () => { + const point = expected({ isl: 1024, joules_per_output_token: 2.5 }); + expect(point.identity).toMatchObject({ benchmark_type: 'single_turn', isl: 1024, osl: 1024 }); + expect(verifyPowerPublication([point], [actual(point)], 'database')).toEqual([]); + expect(verifyPowerPublication([point], [], 'public API')[0]).toContain('found 0'); + }); it('verifies AgentX source identity, nullable sequences, energy and audit through DB/API', () => { const point = expected({ scenario_type: 'agentic-coding', @@ -134,3 +140,47 @@ describe('PowerX publication', () => { expect(tracker.skips.failedRun).toBe(1); }); }); + +describe('NVL72 CPU publication', () => { + it('retains CPU watts and socket provenance and rejects their loss in DB/API readback', () => { + const point = expected({ + cpu_power_valid: 1, + avg_total_cpu_power_w: 501, + power_audit: { + ...raw.power_audit, + cpu: { + sensor_kind: 'grace_socket', + source: 'acpi', + expected_sockets: 2, + observed_sockets: 2, + reason_codes: [], + }, + }, + }); + expect(point.metrics).toMatchObject({ cpu_power_valid: 1, avg_total_cpu_power_w: 501 }); + expect(point.power_audit).toMatchObject({ + cpu: { sensor_kind: 'grace_socket', observed_sockets: 2 }, + }); + expect(verifyPowerPublication([point], [actual(point)], 'database')).toEqual([]); + const missingCpu = actual(point); + delete missingCpu.metrics.cpu_power_valid; + delete missingCpu.metrics.avg_total_cpu_power_w; + expect(verifyPowerPublication([point], [missingCpu], 'API')).toContainEqual( + expect.stringContaining('cpu_power_valid expected 1'), + ); + expect( + verifyPowerPublication([point], [{ ...actual(point), power_audit: raw.power_audit }], 'API'), + ).toContainEqual(expect.stringContaining('power_audit differs')); + }); + + it('preserves valid GPU watts and rejects leaked invalid CPU watts', () => { + const point = expected({ cpu_power_valid: 0, avg_total_cpu_power_w: 501 }); + expect(point.metrics).toMatchObject({ cpu_power_valid: 0, avg_power_w: 642 }); + expect(point.metrics).not.toHaveProperty('avg_total_cpu_power_w'); + const leaked = actual(point); + leaked.metrics.avg_total_cpu_power_w = 501; + expect(verifyPowerPublication([point], [leaked], 'API')).toContainEqual( + expect.stringContaining('avg_total_cpu_power_w expected absent, got 501'), + ); + }); +}); diff --git a/packages/db/src/etl/power-publication.ts b/packages/db/src/etl/power-publication.ts index d2ed10c14..154f5a265 100644 --- a/packages/db/src/etl/power-publication.ts +++ b/packages/db/src/etl/power-publication.ts @@ -1,6 +1,7 @@ import { isDeepStrictEqual } from 'node:util'; import { MEASURED_POWER_METRIC_KEYS } from '@semianalysisai/inferencex-constants'; import type { BenchmarkParams } from './benchmark-mapper'; +import type { TelemetryReceipt } from './telemetry-receipt'; const CONFIG_FIELDS = { hardware: 'hardware', @@ -32,7 +33,12 @@ const IDENTITY_FIELDS = [ 'image', 'run_url', ] as const; -const POWER_FIELDS = [...MEASURED_POWER_METRIC_KEYS, 'power_valid', 'power_metric_schema_version']; +const POWER_FIELDS = [ + ...MEASURED_POWER_METRIC_KEYS, + 'power_valid', + 'power_metric_schema_version', + 'cpu_power_valid', +]; export interface PowerPublicationPoint { identity: Record; metrics: Record; @@ -46,26 +52,67 @@ export interface PowerPublicationManifest { runId: number; runAttempt: number; points: PowerPublicationPoint[]; + /** Fatal: verify-power-publication exits non-zero when this is non-empty. */ ingestErrors?: string[]; + /** + * Non-fatal: PowerX telemetry digest failures. Surfaced in the verification + * receipt so they stay visible, but they never fail the ingest — the benchmark + * rows landed, and the artifact can be re-digested by the backfill. + */ + telemetryWarnings?: string[]; + /** Attachment completeness, separate from benchmark/power publication validity. */ + telemetry?: TelemetryReceipt; + /** Durable refresh responsibility when telemetry recovery fills benchmark metadata. */ + benchmarkRefresh?: { + status: 'pending' | 'complete' | 'failed'; + benchmarkResultIds: number[]; + /** Expected enrichment comes from retained validation, never a DB snapshot. */ + auditUpdates?: { + benchmarkResultId: number; + identity: Record; + /** Original mapped identity when the existing historical offload resolver used a fallback. */ + sourceIdentity?: Record; + powerAudit: { source: string; window_start_unix: number; window_end_unix: number }; + }[]; + endpoint?: string; + checkedAt?: string; + error?: string; + }; +} +/** + * The errors that fail an ingest. `telemetryWarnings` is deliberately not among + * them: a gpu_metrics digest failure costs one point's PowerX tab, while the + * benchmark rows it accompanies are already committed and the artifact can be + * re-digested by `admin:db:backfill-gpu-metrics`. Folding it in would let one + * malformed CSV turn a whole production ingest red. + */ +export function fatalPublicationErrors( + manifest: Pick, + verificationErrors: readonly string[], +): string[] { + return [...(manifest.ingestErrors ?? []), ...verificationErrors]; } + export interface PublishedPowerRow extends Record { metrics: Record; } +export function stablePowerPointIdentity(row: Record): string { + return JSON.stringify( + IDENTITY_FIELDS.filter((key) => key !== 'image' && key !== 'run_url').map( + (key) => row[key] ?? null, + ), + ); +} + export function publicationIdentity(row: Record): string { return JSON.stringify(IDENTITY_FIELDS.map((key) => row[key] ?? null)); } -export function powerPublicationPoint( +export function benchmarkPublicationIdentity( row: BenchmarkParams, - runUrl: string, - artifact: PowerPublicationPoint['artifact'], -): PowerPublicationPoint | null { - if ( - row.benchmarkType !== 'agentic_traces' && - (row.benchmarkType !== 'single_turn' || row.isl !== 8192 || row.osl !== 1024) - ) - return null; + runUrl = '', +): Record { const identity: Record = Object.fromEntries( Object.entries(CONFIG_FIELDS).map(([source, target]) => [ target, @@ -82,6 +129,22 @@ export function powerPublicationPoint( image: row.image, run_url: runUrl, }); + return identity; +} + +export function powerPublicationPoint( + row: BenchmarkParams, + runUrl: string, + artifact: PowerPublicationPoint['artifact'], +): PowerPublicationPoint | null { + if ( + row.benchmarkType !== 'agentic_traces' && + (row.benchmarkType !== 'single_turn' || + (row.isl !== 1024 && row.isl !== 8192) || + row.osl !== 1024) + ) + return null; + const identity = benchmarkPublicationIdentity(row, runUrl); return { identity, metrics: Object.fromEntries( diff --git a/packages/db/src/etl/required-power-curve-db.test.ts b/packages/db/src/etl/required-power-curve-db.test.ts new file mode 100644 index 000000000..7ffcbfb06 --- /dev/null +++ b/packages/db/src/etl/required-power-curve-db.test.ts @@ -0,0 +1,89 @@ +import { PGlite } from '@electric-sql/pglite'; +import fs from 'node:fs'; +import path from 'node:path'; +import os from 'node:os'; +import { beforeAll, beforeEach, afterAll, describe, it, expect } from 'vitest'; +import type { DbClient } from '../connection'; +import { preflightRequiredPowerCurves } from './required-power-curve'; +import { getLatestBenchmarks } from '../queries/benchmarks'; +let db: PGlite; +const sql: DbClient = async (strings, ...values) => { + const query = strings.reduce((text, part, index) => text + (index ? `$${index}` : '') + part, ''); + const result = await db.query>(query, values); + return result.rows; +}; +const golden = path.resolve( + import.meta.dirname, + '../../../../docs/fixtures/powerx-manifest-v2/artifacts', +); +const source = { runId: 123, runAttempt: 1, headSha: 'b'.repeat(40) }; +const options = { date: '2026-09-16', runStartedAt: '2026-09-16T00:00:00Z', appendOnly: false }; +beforeAll(async () => { + db = await PGlite.create(); + const dir = new URL('../../migrations/', import.meta.url); + for (const file of fs + .readdirSync(dir) + .filter((name) => name.endsWith('.sql')) + .sort()) + await db.exec(fs.readFileSync(new URL(file, dir), 'utf8')); +}, 20000); +afterAll(async () => { + await db?.close(); +}); +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + prefill_tp, prefill_ep, decode_tp, decode_ep, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'qwen3.5', 'h100', 'sglang', 'fp8', 'none', false, 1,1,1,1,1,1); + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, date, created_at, run_started_at) + VALUES (1, 111, 1, 'previous', '2026-09-15', '2026-09-15T00:00:00Z', '2026-09-15T00:00:00Z');`); +}); +async function addPoint(conc: number, image = 'example/serving:golden') { + await sql`INSERT INTO benchmark_results (config_id, workflow_run_id, date, benchmark_type, isl, osl, conc, offload_mode, recipe_fingerprint, image, metrics) + VALUES (1,1,'2026-09-15','agentic_traces',NULL,NULL,${conc},'off',${'a'.repeat(64)},${image},'{}')`; +} +describe('read-only required-power DB preflight', () => { + it('rejects partial refresh from base tables before any write, even with a stale latest view', async () => { + await addPoint(1); + await addPoint(64); + const before = await sql`SELECT count(*)::int AS n FROM benchmark_results`; + await expect(preflightRequiredPowerCurves(sql, golden, source, options)).rejects.toThrow( + 'shrink', + ); + expect(await sql`SELECT count(*)::int AS n FROM benchmark_results`).toEqual(before); + expect(await sql`SELECT count(*)::int AS n FROM workflow_runs`).toEqual([{ n: 1 }]); + const published = await getLatestBenchmarks(sql, 'qwen3.5', '9999-12-31'); + expect(published.map((row) => row.conc)).toEqual([1, 64]); + }); + it('detects retry removal from a scope omitted entirely by incoming artifacts', async () => { + await sql`UPDATE workflow_runs SET github_run_id=123`; + await sql`UPDATE configs SET hardware='h200'`; + await addPoint(64); + await expect( + preflightRequiredPowerCurves(sql, golden, { ...source, runAttempt: 2 }, options), + ).rejects.toThrow('shrink'); + expect(await sql`SELECT count(*)::int AS n FROM workflow_runs`).toEqual([{ n: 1 }]); + }); + it('protects optional fixed workloads outside the power receipt whitelist', async () => { + await sql`INSERT INTO benchmark_results (config_id,workflow_run_id,date,benchmark_type,isl,osl,conc,offload_mode,recipe_fingerprint,image,metrics) + VALUES (1,1,'2026-09-15','single_turn',4096,1024,1,'off',${'c'.repeat(64)},'example/serving:golden','{}'), + (1,1,'2026-09-15','single_turn',4096,1024,8,'off',${'c'.repeat(64)},'example/serving:golden','{}')`; + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'powerx-optional-')); + try { + fs.cpSync(golden, dir, { recursive: true }); + fs.mkdirSync(path.join(dir, 'results_optional')); + const row = JSON.parse( + fs.readFileSync(path.join(golden, 'bmk_agentic_golden/agg.json'), 'utf8'), + ); + delete row.scenario_type; + delete row.users; + Object.assign(row, { isl: 4096, osl: 1024, recipe_fingerprint: 'c'.repeat(64) }); + fs.writeFileSync(path.join(dir, 'results_optional/extra.json'), JSON.stringify(row)); + await expect(preflightRequiredPowerCurves(sql, dir, source, options)).rejects.toThrow( + 'shrink', + ); + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); +}); diff --git a/packages/db/src/etl/required-power-curve.test.ts b/packages/db/src/etl/required-power-curve.test.ts new file mode 100644 index 000000000..8dafa5033 --- /dev/null +++ b/packages/db/src/etl/required-power-curve.test.ts @@ -0,0 +1,107 @@ +import { describe, expect, it } from 'vitest'; +import { + assertCurvePreserved, + publishedCurve, + type CurvePoint, + type CurvePublication, +} from './required-power-curve'; +import { powerPublicationPoint, stablePowerPointIdentity } from './power-publication'; +import { verifyRequiredPowerArtifacts } from './required-power-publication'; +import path from 'node:path'; +const golden = path.resolve( + import.meta.dirname, + '../../../../docs/fixtures/powerx-manifest-v2/artifacts', +); +const benchmark = verifyRequiredPowerArtifacts(golden, { + runId: 123, + runAttempt: 1, + headSha: 'b'.repeat(40), +})[0]; +const identity = powerPublicationPoint(benchmark, '', { path: '', sha256: '' })!.identity; +const policy: CurvePublication = { mode: 'incremental', replacement_scope: [] }; +function point(run: number, conc: number, extra: Partial = {}): CurvePoint { + return { + identity: { ...identity, conc }, + image: 'same-image', + workflowRunId: run, + githubRunId: run, + runAttempt: 1, + date: `2026-09-${run.toString().padStart(2, '0')}`, + runStartedAt: null, + appendOnly: false, + ...extra, + }; +} +describe('pure projected publication curve', () => { + it('rejects the old partial-sweep regression without altering input rows', () => { + const old = [ + point(1, 1), + point(1, 16), + point(1, 64, { + identity: { ...identity, conc: 64, disagg: true, recipe_fingerprint: 'split-recipe' }, + }), + ]; + const proposed = [...old, point(2, 1)]; + expect(() => assertCurvePreserved(old, proposed, policy)).toThrow('shrink'); + expect([...publishedCurve(old).values()][0]).toHaveLength(3); + }); + it('does not inherit append-only history with an incompatible image', () => { + const old = [point(1, 1), point(1, 16)]; + expect(() => + assertCurvePreserved( + old, + [...old, point(2, 32, { appendOnly: true, image: 'new-image' })], + policy, + ), + ).toThrow('shrink'); + }); + it('inherits only an uninterrupted same-image append-only chain', () => { + const old = [point(1, 1), point(1, 16), point(2, 32, { appendOnly: true })]; + expect([...publishedCurve(old).values()][0].map((row) => row.identity.conc)).toEqual([ + 32, 1, 16, + ]); + expect(() => + assertCurvePreserved(old, [...old, point(3, 64, { appendOnly: true })], policy), + ).not.toThrow(); + expect(() => assertCurvePreserved(old, [...old, point(3, 64)], policy)).toThrow('shrink'); + }); + it('permits only the exact removed identities of the observed snapshot', () => { + const old = [point(1, 1), point(1, 16)]; + const proposed = [...old, point(2, 1)]; + const replacement: CurvePublication = { + mode: 'replacement', + replacement_scope: [ + { + curve_scope: [...publishedCurve(old).keys()][0], + previous_snapshot_workflow_run_id: 1, + removed_point_identities: [stablePowerPointIdentity(old[1].identity)], + }, + ], + }; + expect(() => assertCurvePreserved(old, proposed, replacement)).not.toThrow(); + for (const altered of [ + { ...replacement.replacement_scope[0], previous_snapshot_workflow_run_id: 99 }, + { ...replacement.replacement_scope[0], removed_point_identities: ['*'] }, + { ...replacement.replacement_scope[0], curve_scope: '*' }, + ]) + expect(() => + assertCurvePreserved(old, proposed, { mode: 'replacement', replacement_scope: [altered] }), + ).toThrow('shrink'); + }); + it('treats topology and offload as point identity inside one AgentX curve', () => { + const old = [point(1, 1)]; + for (const altered of [ + { prefill_tp: 2 }, + { offload_mode: 'on' }, + { recipe_fingerprint: 'other-recipe' }, + ]) { + expect(() => + assertCurvePreserved( + old, + [...old, point(2, 1, { identity: { ...identity, ...altered } })], + policy, + ), + ).toThrow('shrink'); + } + }); +}); diff --git a/packages/db/src/etl/required-power-curve.ts b/packages/db/src/etl/required-power-curve.ts new file mode 100644 index 000000000..1b2cdfbfa --- /dev/null +++ b/packages/db/src/etl/required-power-curve.ts @@ -0,0 +1,287 @@ +import fs from 'node:fs'; +import path from 'node:path'; +import { + benchmarkCurveScope, + type BenchmarkCurveInput, +} from '@semianalysisai/inferencex-constants'; +import type { DbClient } from '../connection'; +import type { Sql } from './db-utils'; +import { REQUIRED_POWER_MANIFEST } from '../lib/ci-artifact-preparation'; +import { mapBenchmarkRow, type BenchmarkParams } from './benchmark-mapper'; +import { configCacheKey, type ConfigParams } from './config-cache'; +import { benchmarkPublicationIdentity, stablePowerPointIdentity } from './power-publication'; +import { + assertRequiredPowerPointsRetained, + verifyRequiredPowerArtifacts, + type RequiredPowerSource, +} from './required-power-publication'; +import { + applyBenchmarkPointBackfill, + isBenchmarkPointPurged, + recordBackfilledPointIdentity, +} from './run-overrides'; +import { createSkipTracker } from './skip-tracker'; + +export interface CurvePoint { + identity: Record; + image: string | null; + workflowRunId: number; + githubRunId: number; + runAttempt: number; + date: string; + runStartedAt: string | null; + appendOnly: boolean; +} +export interface CurveReplacement { + curve_scope: string; + previous_snapshot_workflow_run_id: number; + removed_point_identities: string[]; +} +export interface CurvePublication { + mode: 'incremental' | 'replacement'; + replacement_scope: CurveReplacement[]; +} + +function scope(point: CurvePoint): string { + return benchmarkCurveScope(point.identity as unknown as BenchmarkCurveInput); +} + +/** Mirrors migration 014: latest attempt, scope-level snapshots and same-image append chains. */ +export function publishedCurve(points: readonly CurvePoint[]): Map { + const latestAttempts = new Map(); + for (const point of points) + latestAttempts.set( + point.githubRunId, + Math.max(latestAttempts.get(point.githubRunId) ?? 0, point.runAttempt), + ); + const scopes = new Map>(); + for (const point of points) { + if (point.runAttempt !== latestAttempts.get(point.githubRunId)) continue; + const key = scope(point); + const runs = scopes.get(key) ?? new Map(); + const runDate = JSON.stringify([point.workflowRunId, point.date]); + runs.set(runDate, [...(runs.get(runDate) ?? []), point]); + scopes.set(key, runs); + } + const result = new Map(); + for (const [key, runs] of scopes) { + const ranked = [...runs.values()].sort( + (a, b) => + b[0].date.localeCompare(a[0].date) || + (b[0].runStartedAt ? Date.parse(b[0].runStartedAt) : -Infinity) - + (a[0].runStartedAt ? Date.parse(a[0].runStartedAt) : -Infinity) || + b[0].workflowRunId - a[0].workflowRunId, + ); + const rootImage = ranked[0][0].image; + const uniformImage = (rows: CurvePoint[]) => + rootImage !== null && rows.every((row) => row.image === rootImage); + const selected = new Map(); + for (let index = 0; index < ranked.length; index++) { + const current = ranked[index]; + // SQL groups run/date for ranking, then joins every point in that run/scope. + for (const point of [...runs.values()] + .flat() + .filter((candidate) => candidate.workflowRunId === current[0].workflowRunId)) { + const id = stablePowerPointIdentity(point.identity); + if (!selected.has(id)) selected.set(id, point); + } + if ( + !current[0].appendOnly || + !uniformImage(current) || + !ranked[index + 1] || + !uniformImage(ranked[index + 1]) + ) + break; + } + result.set(key, [...selected.values()]); + } + return result; +} + +function canonical(entries: CurveReplacement[]): string { + return JSON.stringify( + entries + .map((entry) => { + if ( + typeof entry.curve_scope !== 'string' || + !Number.isSafeInteger(entry.previous_snapshot_workflow_run_id) || + !Array.isArray(entry.removed_point_identities) || + entry.removed_point_identities.some((id) => typeof id !== 'string') + ) + throw new Error('Required power: invalid exact replacement scope'); + return { ...entry, removed_point_identities: [...entry.removed_point_identities].sort() }; + }) + .sort((a, b) => a.curve_scope.localeCompare(b.curve_scope)), + ); +} + +/** Exact lost identities bind any destructive authorization to one observed snapshot. */ +export function assertCurvePreserved( + existing: readonly CurvePoint[], + proposed: readonly CurvePoint[], + publication: CurvePublication, +): void { + if ( + !publication || + !['incremental', 'replacement'].includes(publication.mode) || + !Array.isArray(publication.replacement_scope) + ) + throw new Error('Required power: invalid curve publication policy'); + const before = publishedCurve(existing); + const after = publishedCurve(proposed); + const losses: CurveReplacement[] = []; + for (const [key, oldPoints] of before) { + const next = new Set( + (after.get(key) ?? []).map((point) => stablePowerPointIdentity(point.identity)), + ); + const removed = oldPoints + .map((point) => stablePowerPointIdentity(point.identity)) + .filter((identity) => !next.has(identity)) + .sort(); + if (removed.length > 0) + losses.push({ + curve_scope: key, + previous_snapshot_workflow_run_id: oldPoints[0].workflowRunId, + removed_point_identities: removed, + }); + } + if (publication.mode === 'incremental' && publication.replacement_scope.length > 0) + throw new Error('Required power: incremental publication cannot authorize replacement'); + if ( + (losses.length > 0 && publication.mode !== 'replacement') || + canonical(losses) !== canonical(publication.replacement_scope) + ) + throw new Error( + `Required power: publication would shrink the existing curve; exact replacement_scope required: ${JSON.stringify(losses)}`, + ); +} + +/** Read-only preflight; no config/workflow upsert, migration, or materialized-view refresh. */ +export async function preflightRequiredPowerCurves( + sql: DbClient | Sql, + root: string, + source: RequiredPowerSource, + options: { date: string; runStartedAt: string | null; appendOnly: boolean }, +): Promise { + const required = verifyRequiredPowerArtifacts(root, source); + if (required.length === 0) return; + const manifest = JSON.parse( + fs.readFileSync(path.join(root, REQUIRED_POWER_MANIFEST, 'sweep_manifest.json'), 'utf8'), + ); + const configs = await sql`SELECT * FROM configs`; + const configIds = new Map( + configs.map((row) => [ + configCacheKey({ + hardware: row.hardware, + framework: row.framework, + model: row.model, + precision: row.precision, + specMethod: row.spec_method, + disagg: row.disagg, + isMultinode: row.is_multinode, + prefillTp: row.prefill_tp, + prefillEp: row.prefill_ep, + prefillDpAttn: row.prefill_dp_attention, + prefillNumWorkers: row.prefill_num_workers, + decodeTp: row.decode_tp, + decodeEp: row.decode_ep, + decodeDpAttn: row.decode_dp_attention, + decodeNumWorkers: row.decode_num_workers, + numPrefillGpu: row.num_prefill_gpu, + numDecodeGpu: row.num_decode_gpu, + } as ConfigParams), + Number(row.id), + ]), + ); + const incoming = new Map(); + const backfilled = new Map(); + for (const name of fs.readdirSync(root)) { + if ( + (!name.startsWith('bmk_') && !name.startsWith('results_')) || + !fs.statSync(path.join(root, name)).isDirectory() + ) + continue; + for (const file of fs + .readdirSync(path.join(root, name)) + .filter((candidateName) => candidateName.endsWith('.json'))) { + const data = JSON.parse(fs.readFileSync(path.join(root, name, file), 'utf8')); + for (const raw of Array.isArray(data) ? data : [data]) { + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) continue; + const mapped = mapBenchmarkRow(raw, createSkipTracker(), undefined, source.runId); + if (!mapped) continue; + // A config not present yet cannot match a config-id-scoped purge/backfill. + const point = { ...mapped, configId: configIds.get(configCacheKey(mapped.config)) ?? -1 }; + if (isBenchmarkPointPurged(source.runId, source.runAttempt, point)) continue; + const applied = applyBenchmarkPointBackfill(source.runId, source.runAttempt, point); + recordBackfilledPointIdentity(backfilled, applied.sourceIdentity, applied.desiredIdentity); + incoming.set( + stablePowerPointIdentity(benchmarkPublicationIdentity(applied.point)), + applied.point, + ); + } + } + } + assertRequiredPowerPointsRetained(required, [...incoming.values()]); + const models = [...new Set([...incoming.values()].map((point) => point.config.model))]; + const rows = await sql` + SELECT c.*, br.benchmark_type, br.isl, br.osl, br.conc, br.offload_mode, + br.recipe_fingerprint, br.image, br.date::text AS point_date, + wr.id AS workflow_run_id, wr.github_run_id, wr.run_attempt, + wr.run_started_at::text, wr.append_only + FROM benchmark_results br + JOIN configs c ON c.id = br.config_id + JOIN latest_workflow_runs wr ON wr.id = br.workflow_run_id + WHERE br.error IS NULL AND (c.model = ANY(${models}) OR wr.github_run_id = ${source.runId})`; + const stored = rows.map((row): CurvePoint => ({ + identity: row, + image: row.image as string | null, + workflowRunId: Number(row.workflow_run_id), + githubRunId: Number(row.github_run_id), + runAttempt: Number(row.run_attempt), + date: String(row.point_date), + runStartedAt: row.run_started_at as string | null, + appendOnly: Boolean(row.append_only), + })); + const attempts = + await sql`SELECT id, run_attempt FROM workflow_runs WHERE github_run_id = ${source.runId}`; + const sameAttempt = attempts.find((row) => Number(row.run_attempt) === source.runAttempt); + // New serial IDs sort after stored IDs. A concurrent ingest is outside this pure preflight boundary. + const workflowRunId = sameAttempt ? Number(sameAttempt.id) : Number.MAX_SAFE_INTEGER; + const newPoints = [...incoming.values()].map((point): CurvePoint => ({ + identity: benchmarkPublicationIdentity(point), + image: point.image, + workflowRunId, + githubRunId: source.runId, + runAttempt: source.runAttempt, + date: + stored.find( + (existing) => + existing.workflowRunId === workflowRunId && + stablePowerPointIdentity(existing.identity) === + stablePowerPointIdentity(benchmarkPublicationIdentity(point)), + )?.date ?? options.date, + runStartedAt: options.runStartedAt, + appendOnly: options.appendOnly, + })); + const touched = new Set( + [...newPoints, ...stored.filter((point) => point.githubRunId === source.runId)].map(scope), + ); + const existing = stored.filter((point) => touched.has(scope(point))); + const maxAttempt = Math.max(source.runAttempt, ...attempts.map((row) => Number(row.run_attempt))); + const changedIds = new Set(newPoints.map((point) => stablePowerPointIdentity(point.identity))); + const proposed = existing + .filter( + (point) => + point.githubRunId !== source.runId || + (point.runAttempt === maxAttempt && + (source.runAttempt !== maxAttempt || + !changedIds.has(stablePowerPointIdentity(point.identity)))), + ) + .map((point) => + point.githubRunId === source.runId && point.runAttempt === source.runAttempt + ? { ...point, runStartedAt: options.runStartedAt, appendOnly: options.appendOnly } + : point, + ); + if (source.runAttempt === maxAttempt) proposed.push(...newPoints); + assertCurvePreserved(existing, proposed, manifest.publication); +} diff --git a/packages/db/src/etl/required-power-publication.test.ts b/packages/db/src/etl/required-power-publication.test.ts index b51b8e526..ec1822234 100644 --- a/packages/db/src/etl/required-power-publication.test.ts +++ b/packages/db/src/etl/required-power-publication.test.ts @@ -1,314 +1,200 @@ +import { createHash } from 'node:crypto'; import fs from 'node:fs'; import os from 'node:os'; import path from 'node:path'; import { afterEach, describe, expect, it } from 'vitest'; -import agenticMatrix from './__fixtures__/required-power-agentic-matrix.json'; import { assertRequiredPowerPointsRetained, verifyRequiredPowerArtifacts, - verifyRequiredPowerPublication, } from './required-power-publication'; -import { - isBenchmarkPointPurged, - PURGED_BENCHMARK_POINTS, - type PurgedBenchmarkPoint, -} from './run-overrides'; -// Ordinary planner output for Qwen3.5/H100 8K1K; fingerprints include the full recipe. -const fingerprint = 'fa0645b4dec181eafda7472c891e90d9a5bf7bad7ee764aa0fdcc41ff841adf1'; -const source = { runId: 123, runAttempt: 2, headSha: 'abc123' }; -const required = { - 'recipe-fingerprint': fingerprint, - conc: 1, - isl: 8192, - osl: 1024, - 'require-power': true, -}; -const row = { - infmax_model_prefix: 'qwen3.5', - hw: 'h100', - framework: 'sglang', - precision: 'fp8', - recipe_fingerprint: fingerprint, - conc: 1, - isl: 8192, - osl: 1024, - tp: 8, - ep: 1, - power_valid: 1, - power_metric_schema_version: 2, - avg_power_w: 500, - avg_total_gpu_power_w: 4000, - total_gpu_energy_j: 8000, - joules_per_output_token: 2, -}; -const manifest = (entries = [required]) => ({ - head: source.headSha, - 'run-id': source.runId, - 'run-attempt': source.runAttempt, - // full-sweep is a Klaud policy field: fail-fast runs may legitimately set it false. - 'full-sweep': false, - matrix: { - single_node: { '8k1k': entries }, - multi_node: {}, - evals: [{ ...required, 'eval-only': true }], - }, -}); -const artifacts = (rows: unknown[] = [row]) => [{ path: 'bmk_qwen/agg.json', rows }]; +const golden = path.resolve( + import.meta.dirname, + '../../../../docs/fixtures/powerx-manifest-v2/artifacts', +); +const source = { runId: 123, runAttempt: 1, headSha: 'b'.repeat(40) }; const dirs: string[] = []; +function fixture() { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'powerx-contract-')); + dirs.push(dir); + fs.cpSync(golden, dir, { recursive: true }); + return dir; +} +function json(dir: string, file: string): any { + return JSON.parse(fs.readFileSync(path.join(dir, file), 'utf8')); +} +function write(dir: string, file: string, value: unknown) { + fs.writeFileSync(path.join(dir, file), JSON.stringify(value)); +} +const manifestPath = 'required-power-sweep-manifest/sweep_manifest.json'; +const benchmarkPath = 'bmk_agentic_golden/agg.json'; +const auditPath = 'agentic_golden/power_validation.json'; +function changeManifest(dir: string, edit: (manifest: any) => void) { + const manifest = json(dir, manifestPath); + edit(manifest); + write(dir, manifestPath, manifest); +} +function changeArtifact(dir: string, file: string, edit: (value: any) => void) { + const value = json(dir, file); + edit(value); + write(dir, file, value); + changeManifest(dir, (manifest) => { + for (const point of manifest.points) + for (const artifact of point.artifacts) + if (artifact.path === file) + artifact.sha256 = createHash('sha256') + .update(fs.readFileSync(path.join(dir, file))) + .digest('hex'); + }); +} +function multinodeFixture() { + const dir = fixture(); + const extras: Record = { + 'power_audit_golden/power/samples.csv': + 'timestamp_unix,hostname,gpu_uuid,power_w\n1700000000,prefill,GPU-p,200\n', + 'power_audit_golden/power/manifest.json': { status: 'complete', publication_valid: true }, + 'power_audit_golden/power/windows/point.json': { + benchmark_start_time_unix: 1700000000, + benchmark_end_time_unix: 1700000002, + }, + }; + for (const [file, value] of Object.entries(extras)) { + fs.mkdirSync(path.dirname(path.join(dir, file)), { recursive: true }); + fs.writeFileSync( + path.join(dir, file), + typeof value === 'string' ? value : JSON.stringify(value), + ); + } + changeArtifact(dir, benchmarkPath, (row) => + Object.assign(row, { + disagg: true, + is_multinode: true, + num_gpus: 2, + num_prefill_gpu: 1, + num_decode_gpu: 1, + prefill_gpu_energy_j: 400, + decode_gpu_energy_j: 600, + prefill_joules_per_input_token: 1, + decode_joules_per_output_token: 1, + }), + ); + changeArtifact(dir, auditPath, (audit) => + Object.assign(audit, { + expected_gpu_count: 2, + observed_gpu_count: 2, + per_gpu_energy_j: { 'prefill/GPU-p': 400, 'decode/GPU-d': 600 }, + per_gpu_role: { 'prefill/GPU-p': 'prefill', 'decode/GPU-d': 'decode' }, + }), + ); + changeManifest(dir, (manifest) => { + const entry = manifest.matrix.single_node.agentic[0]; + Object.assign(entry, { disagg: true, 'num-gpus': 2, 'node-count': 2 }); + manifest.matrix = { single_node: {}, multi_node: { agentic: [entry] } }; + const point = manifest.points[0]; + Object.assign(point.topology, { + disagg: true, + is_multinode: true, + num_gpus: 2, + num_prefill_gpu: 1, + num_decode_gpu: 1, + }); + point.devices = [ + { node: 'prefill', gpu_uuid: 'GPU-p', role: 'prefill', energy_j: 400 }, + { node: 'decode', gpu_uuid: 'GPU-d', role: 'decode', energy_j: 600 }, + ]; + for (const file of Object.keys(extras)) + point.artifacts.push({ + path: file, + sha256: createHash('sha256') + .update(fs.readFileSync(path.join(dir, file))) + .digest('hex'), + validation_state: 'valid', + }); + }); + return dir; +} afterEach(() => { for (const dir of dirs.splice(0)) fs.rmSync(dir, { recursive: true, force: true }); }); -describe('required power publication preflight', () => { - it('matches the canonical AgentX users value instead of a conflicting raw conc', () => { - const scope = { - ...manifest(), - matrix: { single_node: { agentic: [required] }, multi_node: {} }, - }; - const conflicting = { ...row, scenario_type: 'agentic-coding', conc: 1, users: 2 }; - expect(() => verifyRequiredPowerPublication(scope, artifacts([conflicting]), source)).toThrow( - 'missing benchmark point', - ); - expect( - verifyRequiredPowerPublication( - scope, - artifacts([{ ...conflicting, conc: 2, users: 1 }]), - source, - ), - ).toHaveLength(1); +describe('required power publication contract', () => { + it('fails missing required manifests even when changelog metadata is also missing', () => { + const dir = fixture(); + fs.rmSync(path.join(dir, 'required-power-sweep-manifest'), { recursive: true }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('sweep manifest missing'); + fs.rmSync(path.join(dir, 'changelog-metadata'), { recursive: true }); + expect(() => verifyRequiredPowerArtifacts(dir, source, true)).toThrow('sweep manifest missing'); + expect(verifyRequiredPowerArtifacts(dir, source)).toEqual([]); }); - - it('accepts exact ordinary source points, including fail-fast manifests and identical aggregate copies', () => { - expect( - verifyRequiredPowerPublication( - manifest(), - [...artifacts(), { path: 'results_bmk/agg_bmk.json', rows: [{ ...row }] }], - source, - ), - ).toHaveLength(1); + it('rejects hash mismatches and explicit invalid evidence', () => { + const dir = fixture(); + fs.appendFileSync(path.join(dir, 'agentic_golden/gpu_metrics.csv'), '\n'); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('hash'); + const invalid = fixture(); + changeManifest(invalid, (manifest) => { + manifest.points[0].artifacts[0].validation_state = 'invalid'; + }); + expect(() => verifyRequiredPowerArtifacts(invalid, source)).toThrow('validation'); }); - - it('fails required coverage after a point purge even when an optional point is retained', () => { - const requiredPoints = verifyRequiredPowerPublication(manifest(), artifacts(), source); - const candidates = [requiredPoints[0], { ...requiredPoints[0], conc: 2 }]; - const purged: PurgedBenchmarkPoint = { - githubRunId: source.runId, - runAttempt: source.runAttempt, - configId: 999999, - benchmarkType: 'single_turn', - isl: 8192, - osl: 1024, - conc: 1, - offloadMode: 'off', - recipeFingerprint: fingerprint, - }; - const registry = PURGED_BENCHMARK_POINTS as PurgedBenchmarkPoint[]; - registry.push(purged); - try { - const retained = candidates.filter( - (point) => - !isBenchmarkPointPurged(source.runId, source.runAttempt, { - ...point, - configId: purged.configId, - }), - ); - expect(retained.map((point) => point.conc)).toEqual([2]); - expect(() => assertRequiredPowerPointsRetained(requiredPoints, retained)).toThrow( - 'missing benchmark point after ingest', - ); - expect(isBenchmarkPointPurged(source.runId, source.runAttempt, purged)).toBe(true); - expect(() => assertRequiredPowerPointsRetained([], retained)).not.toThrow(); - } finally { - registry.splice(registry.indexOf(purged), 1); - } - }); - - it('accepts retained aggregate copies but cannot replace a filtered required point with another identity', () => { - const requiredPoints = verifyRequiredPowerPublication(manifest(), artifacts(), source); - expect(() => - assertRequiredPowerPointsRetained(requiredPoints, [ - requiredPoints[0], - { ...requiredPoints[0] }, - ]), - ).not.toThrow(); - expect(() => - assertRequiredPowerPointsRetained(requiredPoints, [ - { ...requiredPoints[0], recipeFingerprint: 'other-recipe' }, - ]), - ).toThrow('missing benchmark point after ingest'); - }); - - it('fails before ingestion when any declared concurrency, fingerprint or scenario is missing', () => { - for (const candidate of [ - [], - [{ ...row, conc: 2 }], - [{ ...row, recipe_fingerprint: '' }], - [{ ...row, isl: 1024 }], - ]) { - expect(() => - verifyRequiredPowerPublication(manifest(), artifacts(candidate), source), - ).toThrow('missing benchmark point'); - } - }); - - it('does not retroactively require optional matrix points or legacy results', () => { - const scope = manifest([required, { ...required, conc: 2, 'require-power': false }]); - expect( - verifyRequiredPowerPublication(scope, artifacts([row, { conc: 2, power_valid: 0 }]), source), - ).toHaveLength(1); - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'required-power-')); - dirs.push(dir); - expect(verifyRequiredPowerArtifacts(dir, { ...source, headSha: null })).toHaveLength(0); + it('rejects a measured zero energy', () => { + const dir = fixture(); + changeArtifact(dir, benchmarkPath, (rows) => { + rows.total_gpu_energy_j = 0; + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow( + 'total_gpu_energy_j must be finite and positive', + ); }); - - it.each([ - { power_valid: 0 }, - { power_valid: '1' }, - { power_metric_schema_version: '2' }, - { avg_power_w: 0 }, - { avg_total_gpu_power_w: -1 }, - { total_gpu_energy_j: NaN }, - { joules_per_output_token: Infinity }, - { joules_per_output_token: undefined }, - { benchmark_outcome: { status: 'failed' } }, - ])('rejects invalid required measurements: %j', (change) => { - expect(() => - verifyRequiredPowerPublication(manifest(), artifacts([{ ...row, ...change }]), source), - ).toThrow('Required power:'); + it('does not replace canonical AgentX users with a conflicting raw conc', () => { + const dir = fixture(); + changeArtifact(dir, benchmarkPath, (rows) => { + rows.conc = 1; + rows.users = 2; + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('missing benchmark point'); }); - - it('rejects conflicting copies and duplicate rows inside either source or aggregate file', () => { - expect(() => - verifyRequiredPowerPublication( - manifest(), - [...artifacts(), { path: 'results_bmk/agg.json', rows: [{ ...row, avg_power_w: 501 }] }], - source, - ), - ).toThrow('conflicting'); - expect(() => verifyRequiredPowerPublication(manifest(), artifacts([row, row]), source)).toThrow( - 'duplicate', + it('requires both prefill and decode physical evidence and matching role energy', () => { + const dir = multinodeFixture(); + expect(verifyRequiredPowerArtifacts(dir, source)).toHaveLength(1); + changeManifest(dir, (manifest) => { + manifest.points[0].devices.pop(); + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('missing physical GPU'); + const mismatched = multinodeFixture(); + changeArtifact(mismatched, benchmarkPath, (row) => { + row.prefill_gpu_energy_j = 100; + }); + expect(() => verifyRequiredPowerArtifacts(mismatched, source)).toThrow( + 'prefill energy differs', ); - expect(() => - verifyRequiredPowerPublication( - manifest(), - [...artifacts(), { path: 'results_bmk/agg.json', rows: [row, row] }], - source, - ), - ).toThrow('duplicate'); }); - - it('rejects stale source run, attempt or head, an empty scope, and duplicate planned points', () => { - for (const changed of [ - { runId: 999 }, - { runAttempt: 1 }, - { headSha: 'stale' }, - { headSha: null }, - ]) - expect(() => - verifyRequiredPowerPublication(manifest(), artifacts(), { ...source, ...changed }), - ).toThrow('manifest source'); - expect(() => verifyRequiredPowerPublication(manifest([]), [], source)).toThrow('no required'); - expect(() => - verifyRequiredPowerPublication(manifest([required, required]), artifacts(), source), - ).toThrow('duplicate matrix'); + it('rejects invalid sidecar verdict and device energy disagreement despite valid hashes', () => { + const dir = fixture(); + changeArtifact(dir, auditPath, (audit) => { + audit.power_valid = false; + }); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('invalid audit'); + const other = fixture(); + changeManifest(other, (manifest) => { + manifest.points[0].devices[0].energy_j = 2; + }); + expect(() => verifyRequiredPowerArtifacts(other, source)).toThrow('differs from audit'); }); - - it('reuses the declared scope from a successful earlier attempt of the same run and head', () => { - expect( - verifyRequiredPowerPublication({ ...manifest(), 'run-attempt': 1 }, artifacts(), source), - ).toHaveLength(1); - }); - - it('expands real ordinary AgentX planner concurrency arrays and fixed-sequence batches', () => { - // First B200 recipe from the ordinary Kimi K3 required-power planner, 2026-09-13. - const entry = agenticMatrix.multi_node.agentic[0]; - expect( - verifyRequiredPowerPublication( - { ...manifest(), matrix: agenticMatrix }, - artifacts([ - { - ...row, - infmax_model_prefix: 'kimik3', - hw: 'b200', - framework: 'dynamo-vllm', - precision: 'fp4', - recipe_fingerprint: entry['recipe-fingerprint'], - conc: entry.conc[0], - scenario_type: 'agentic-coding', - }, - ]), - source, - ), - ).toHaveLength(1); - const fixedMatrix = { - single_node: {}, - multi_node: { '8k1k': [{ ...required, conc: [1, 2, 4] }] }, - }; - expect( - verifyRequiredPowerPublication( - { ...manifest(), matrix: fixedMatrix }, - artifacts([1, 2, 4].map((conc) => ({ ...row, conc }))), - source, - ), - ).toHaveLength(3); - expect(() => - verifyRequiredPowerPublication({ ...manifest(), matrix: fixedMatrix }, artifacts(), source), - ).toThrow('missing benchmark'); + it('accepts identical aggregate copies but rejects conflicting duplicate rows', () => { + const dir = fixture(); + fs.mkdirSync(path.join(dir, 'results_bmk')); + fs.copyFileSync(path.join(dir, benchmarkPath), path.join(dir, 'results_bmk/agg.json')); + expect(verifyRequiredPowerArtifacts(dir, source)).toHaveLength(1); + const rows = json(dir, benchmarkPath); + rows.avg_power_w = 501; + write(dir, 'results_bmk/agg.json', rows); + expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('conflicting'); }); - - it('accepts AgentX and requires both energy roles only for a declared disaggregated recipe', () => { - const scope = manifest(); - const agentic = { ...required, 'scenario-type': 'agentic-coding', disagg: true }; - const agenticManifest = { - ...scope, - matrix: { single_node: {}, multi_node: { agentic: [agentic] } }, - }; - const split = { - ...row, - isl: undefined, - osl: undefined, - scenario_type: 'agentic-coding', - disagg: true, - prefill_gpu_energy_j: 3000, - decode_gpu_energy_j: 5000, - prefill_joules_per_input_token: 1, - decode_joules_per_output_token: 1.25, - }; - expect( - verifyRequiredPowerPublication(agenticManifest, artifacts([split]), source), - ).toHaveLength(1); + it('retains every required identity after local purges and backfills', () => { + const required = verifyRequiredPowerArtifacts(golden, source); expect(() => - verifyRequiredPowerPublication( - agenticManifest, - artifacts([{ ...split, decode_gpu_energy_j: undefined }]), - source, - ), - ).toThrow('decode_gpu_energy_j'); - const aggregate = { - ...agenticManifest, - matrix: { single_node: {}, multi_node: { agentic: [{ ...agentic, disagg: false }] } }, - }; - expect( - verifyRequiredPowerPublication( - aggregate, - artifacts([{ ...row, scenario_type: 'agentic-coding' }]), - source, - ), - ).toHaveLength(1); - }); - - it('reads the producer artifact and checks all rows before the ingest entry point writes', () => { - const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'required-power-')); - dirs.push(dir); - fs.mkdirSync(path.join(dir, 'required-power-sweep-manifest')); - fs.writeFileSync( - path.join(dir, 'required-power-sweep-manifest', 'sweep_manifest.json'), - JSON.stringify(manifest()), - ); - expect(() => verifyRequiredPowerArtifacts(dir, source)).toThrow('missing benchmark'); - fs.mkdirSync(path.join(dir, 'bmk_qwen')); - fs.writeFileSync(path.join(dir, 'bmk_qwen', 'agg.json'), JSON.stringify(row)); - expect(verifyRequiredPowerArtifacts(dir, source)).toHaveLength(1); + assertRequiredPowerPointsRetained(required, [{ ...required[0], conc: 2 }]), + ).toThrow('missing benchmark point'); + expect(() => assertRequiredPowerPointsRetained(required, required)).not.toThrow(); }); }); diff --git a/packages/db/src/etl/required-power-publication.ts b/packages/db/src/etl/required-power-publication.ts index 170b3e926..d8f995b8a 100644 --- a/packages/db/src/etl/required-power-publication.ts +++ b/packages/db/src/etl/required-power-publication.ts @@ -1,9 +1,11 @@ import fs from 'node:fs'; +import { createHash } from 'node:crypto'; import path from 'node:path'; import { isDeepStrictEqual } from 'node:util'; import { mapBenchmarkRow, type BenchmarkParams } from './benchmark-mapper'; import { createSkipTracker } from './skip-tracker'; -import { REQUIRED_POWER_MANIFEST } from '../lib/ci-artifact-preparation'; +import { configCacheKey } from './config-cache'; +import { CHANGELOG_ARTIFACT_NAME, REQUIRED_POWER_MANIFEST } from '../lib/ci-artifact-preparation'; type JsonRow = Record; export interface RequiredPowerSource { @@ -14,6 +16,7 @@ export interface RequiredPowerSource { export interface BenchmarkArtifactRows { path: string; rows: unknown[]; + contents?: Buffer | string; } function object(value: unknown, label: string): JsonRow { @@ -45,11 +48,24 @@ export function assertRequiredPowerPointsRetained( required: readonly BenchmarkParams[], retained: readonly BenchmarkParams[], ): void { - const present = new Set(retained.map(mappedIdentity)); + const retainedIdentity = (row: BenchmarkParams) => + JSON.stringify([configCacheKey(row.config), row.offloadMode, mappedIdentity(row)]); + const present = new Map(retained.map((row) => [retainedIdentity(row), row])); for (const row of required) { - const key = mappedIdentity(row); + const key = retainedIdentity(row); if (!present.has(key)) throw new Error(`Required power: missing benchmark point after ingest ${key}`); + const actual = present.get(key)!; + for (const field of [ + 'power_valid', + 'power_metric_schema_version', + 'avg_power_w', + 'avg_total_gpu_power_w', + 'total_gpu_energy_j', + 'joules_per_output_token', + ]) + if (actual.metrics[field] !== row.metrics[field]) + throw new Error(`Required power: ${field} changed before ingest for ${key}`); } } @@ -60,20 +76,52 @@ export function verifyRequiredPowerPublication( source: RequiredPowerSource, ): BenchmarkParams[] { const manifest = object(manifestValue, 'sweep manifest'); + if (manifest['schema-version'] !== 2) + throw new Error('Required power: incompatible manifest schema-version (expected 2)'); + const publication = object(manifest.publication, 'publication policy'); + if ( + !['incremental', 'replacement'].includes(String(publication.mode)) || + !Array.isArray(publication.replacement_scope) || + (publication.mode === 'incremental' && publication.replacement_scope.length > 0) + ) + throw new Error('Required power: invalid publication policy'); + const replacementScopes = new Set(); + for (const value of publication.replacement_scope) { + const replacement = object(value, 'exact replacement scope'); + if ( + typeof replacement.curve_scope !== 'string' || + !replacement.curve_scope || + replacementScopes.has(replacement.curve_scope) || + !Number.isSafeInteger(replacement.previous_snapshot_workflow_run_id) || + Number(replacement.previous_snapshot_workflow_run_id) <= 0 || + !Array.isArray(replacement.removed_point_identities) || + replacement.removed_point_identities.length === 0 || + replacement.removed_point_identities.some((id) => typeof id !== 'string' || !id) || + new Set(replacement.removed_point_identities).size !== + replacement.removed_point_identities.length + ) + throw new Error('Required power: invalid exact replacement scope'); + replacementScopes.add(replacement.curve_scope); + } const declaredAttempt = manifest['run-attempt']; if ( manifest['run-id'] !== source.runId || + !Number.isSafeInteger(source.runId) || + source.runId <= 0 || + !Number.isSafeInteger(source.runAttempt) || + source.runAttempt <= 0 || typeof declaredAttempt !== 'number' || !Number.isSafeInteger(declaredAttempt) || declaredAttempt <= 0 || declaredAttempt > source.runAttempt || !source.headSha || + !/^[a-f0-9]{40}$/u.test(source.headSha) || manifest.head !== source.headSha ) throw new Error('Required power: manifest source run, attempt or head does not match'); const matrix = object(manifest.matrix, 'sweep matrix'); - const expected = new Map(); + const expected = new Map(); for (const topology of ['single_node', 'multi_node']) { const buckets = object(matrix[topology], topology); for (const [scenario, entries] of Object.entries(buckets)) { @@ -99,7 +147,7 @@ export function verifyRequiredPowerPublication( osl, ); if (expected.has(key)) throw new Error(`Required power: duplicate matrix point ${key}`); - expected.set(key, row.disagg === true); + expected.set(key, { ...row, matrixTopology: topology }); } } } @@ -140,7 +188,7 @@ export function verifyRequiredPowerPublication( 'total_gpu_energy_j', 'joules_per_output_token', ]; - if (expected.get(key)) + if (expected.get(key)?.disagg === true) fields.push( 'prefill_gpu_energy_j', 'decode_gpu_energy_j', @@ -158,6 +206,7 @@ export function verifyRequiredPowerPublication( for (const key of expected.keys()) { if (!seen.has(key)) throw new Error(`Required power: missing benchmark point ${key}`); } + verifyPointEvidence(manifest, expected, seen, artifacts); return [...seen.values()].map(({ point }) => point); } @@ -165,9 +214,21 @@ export function verifyRequiredPowerPublication( export function verifyRequiredPowerArtifacts( root: string, source: RequiredPowerSource, + required = false, ): BenchmarkParams[] { const manifestDir = path.join(root, REQUIRED_POWER_MANIFEST); - if (!fs.existsSync(manifestDir)) return []; + if (!fs.existsSync(manifestDir)) { + if (required) throw new Error('Required power: sweep manifest missing for required dispatch'); + const metadataDir = path.join(root, CHANGELOG_ARTIFACT_NAME); + if (fs.existsSync(metadataDir)) { + for (const name of fs.readdirSync(metadataDir).filter((file) => file.endsWith('.json'))) { + const metadata = JSON.parse(fs.readFileSync(path.join(metadataDir, name), 'utf8')); + if (metadata?.['require-power'] === true) + throw new Error('Required power: sweep manifest missing for required changelog scope'); + } + } + return []; + } const manifest = JSON.parse( fs.readFileSync(path.join(manifestDir, 'sweep_manifest.json'), 'utf8'), ); @@ -178,8 +239,28 @@ export function verifyRequiredPowerArtifacts( if (!fs.statSync(dir).isDirectory()) continue; for (const file of fs.readdirSync(dir)) { if (!file.endsWith('.json')) continue; - const data = JSON.parse(fs.readFileSync(path.join(dir, file), 'utf8')); - artifacts.push({ path: path.join(name, file), rows: Array.isArray(data) ? data : [data] }); + const contents = fs.readFileSync(path.join(dir, file)); + const data = JSON.parse(contents.toString('utf8')); + artifacts.push({ + path: path.join(name, file), + rows: Array.isArray(data) ? data : [data], + contents, + }); + } + } + for (const pointValue of Array.isArray(manifest.points) ? manifest.points : []) { + const point = object(pointValue, 'point'); + if (!Array.isArray(point.artifacts)) throw new Error('Required power: missing point artifacts'); + for (const value of point.artifacts) { + const artifact = object(value, 'artifact'); + const relative = safeArtifactPath(artifact.path); + const file = path.join(root, relative); + if (!fs.existsSync(file) || !fs.statSync(file).isFile()) + throw new Error(`Required power: missing required artifact ${relative}`); + if (!fs.realpathSync(file).startsWith(`${fs.realpathSync(root)}${path.sep}`)) + throw new Error(`Required power: artifact escapes bundle ${relative}`); + if (!artifacts.some((item) => item.path === relative)) + artifacts.push({ path: relative, rows: [], contents: fs.readFileSync(file) }); } } const points = verifyRequiredPowerPublication(manifest, artifacts, source); @@ -189,3 +270,424 @@ export function verifyRequiredPowerArtifacts( ); return points; } + +function digest(artifact: BenchmarkArtifactRows): string { + return createHash('sha256').update(artifact.contents!).digest('hex'); +} + +function safeArtifactPath(value: unknown): string { + if ( + typeof value !== 'string' || + !value || + value.includes('\\') || + path.posix.isAbsolute(value) || + value.split('/').some((part) => !part || part === '.' || part === '..') + ) + throw new Error('Required power: invalid artifact path'); + return value; +} + +function positive(value: unknown, label: string): number { + if (typeof value !== 'number' || !Number.isFinite(value) || value <= 0) + throw new Error(`Required power: ${label} must be finite and positive`); + return value; +} + +function nonempty(value: unknown, label: string): string { + if (typeof value !== 'string' || !value.trim()) + throw new Error(`Required power: missing ${label}`); + return value; +} + +function verifyPointEvidence( + manifest: JsonRow, + expected: Map, + seen: Map, + artifacts: readonly BenchmarkArtifactRows[], +): void { + if (!Array.isArray(manifest.points) || manifest.points.length !== expected.size) + throw new Error('Required power: point manifest does not cover the exact required matrix'); + const verified = new Set(); + const files = new Map(artifacts.map((artifact) => [artifact.path, artifact])); + for (const declaredValue of manifest.points) { + const declared = object(declaredValue, 'point'); + const id = object(declared.identity, 'point identity'); + const key = identity( + id.recipe_fingerprint, + id.concurrency, + String(id.benchmark_type), + id.isl, + id.osl, + ); + const matched = seen.get(key); + const matrix = expected.get(key); + if (!matched || !matrix || verified.has(key)) + throw new Error(`Required power: unexpected or duplicate manifest point ${key}`); + verified.add(key); + if (nonempty(declared.config_key, 'config_key') !== matrix['exp-name']) + throw new Error(`Required power: config_key differs from required matrix for ${key}`); + const { row, point } = matched; + for (const [field, rawField] of [ + ['model', 'infmax_model_prefix'], + ['hardware', 'hw'], + ['framework', 'framework'], + ['precision', 'precision'], + ]) { + if ( + nonempty(id[field], field) !== + (row[rawField] ?? (field === 'model' ? row.model : undefined)) + ) + throw new Error(`Required power: ${field} identity differs from benchmark for ${key}`); + } + for (const [field, matrixField] of [ + ['model', 'model-prefix'], + ['hardware', 'runner'], + ['framework', 'framework'], + ['precision', 'precision'], + ]) { + if (id[field] !== matrix[matrixField]) + throw new Error( + `Required power: ${field} identity differs from required matrix for ${key}`, + ); + } + const topology = object(declared.topology, 'point topology'); + if ( + topology.disagg !== point.config.disagg || + topology.is_multinode !== point.config.isMultinode || + topology.disagg !== (matrix.disagg === true) || + topology.is_multinode !== (matrix.matrixTopology === 'multi_node') + ) + throw new Error(`Required power: topology differs from benchmark or matrix for ${key}`); + const rawGpuCount = + row.num_gpus ?? + (topology.is_multinode + ? Number(row.num_prefill_gpu) + Number(row.num_decode_gpu) + : Number(row.tp) * Number(row.pp ?? 1) * Number(row.pcp_size ?? 1)); + if (topology.num_gpus !== rawGpuCount) + throw new Error(`Required power: num_gpus topology differs from benchmark for ${key}`); + const topologyFields = [ + 'num_prefill_gpu', + 'num_decode_gpu', + 'tp', + 'ep', + 'dp_attention', + 'prefill_tp', + 'prefill_ep', + 'prefill_num_workers', + 'decode_tp', + 'decode_ep', + 'decode_num_workers', + 'pp', + 'pcp_size', + 'dcp_size', + 'prefill_pp', + 'decode_pp', + 'prefill_pcp_size', + 'decode_pcp_size', + 'prefill_dcp_size', + 'decode_dcp_size', + 'prefill_dp_attention', + 'decode_dp_attention', + ]; + for (const field of topologyFields) { + if (topology[field] !== row[field]) + throw new Error(`Required power: ${field} topology differs from benchmark for ${key}`); + const role = field.startsWith('prefill_') + ? 'prefill' + : field.startsWith('decode_') + ? 'decode' + : null; + const roleField = role ? field.slice(role.length + 1) : field; + const matrixField = + roleField === 'num_workers' + ? 'num-worker' + : roleField === 'dp_attention' + ? 'dp-attn' + : roleField.replaceAll('_', '-'); + let planned = role + ? matrix[role] + ? object(matrix[role], `${role} matrix`)[matrixField] + : matrix[field.replaceAll('_', '-')] + : matrix[matrixField]; + if ( + role === 'decode' && + matrix.decode && + object(matrix.decode, 'decode matrix')['num-worker'] === 0 && + ['tp', 'ep', 'pp', 'pcp_size', 'dcp_size'].includes(roleField) + ) + planned = ['tp', 'ep'].includes(roleField) ? 0 : 1; + let actual = + topology[field] ?? + (['pp', 'pcp_size', 'dcp_size'].includes(roleField) + ? 1 + : roleField === 'dp_attention' + ? false + : undefined); + if (roleField === 'dp_attention' && typeof actual === 'string') actual = actual === 'true'; + const normalizedPlanned = + roleField === 'dp_attention' && typeof planned === 'string' ? planned === 'true' : planned; + if (planned !== undefined && actual !== normalizedPlanned) + throw new Error( + `Required power: ${field} topology differs from required matrix for ${key}`, + ); + } + let plannedGpuCount = matrix['num-gpus']; + if (matrix.prefill && matrix.decode) { + plannedGpuCount = 0; + for (const role of ['prefill', 'decode']) { + const planned = object(matrix[role], `${role} matrix`); + const count = + Number(planned.tp) * + Number(planned.pp ?? 1) * + Number(planned['pcp-size'] ?? 1) * + Number(planned['num-worker']); + if (count !== topology[`num_${role}_gpu`]) + throw new Error( + `Required power: ${role} GPU count differs from required matrix for ${key}`, + ); + plannedGpuCount = Number(plannedGpuCount) + count; + } + } else if (plannedGpuCount === undefined && matrix.tp !== undefined) { + plannedGpuCount = + Number(matrix.tp) * Number(matrix.pp ?? 1) * Number(matrix['pcp-size'] ?? 1); + } + if (plannedGpuCount !== undefined && plannedGpuCount !== topology.num_gpus) + throw new Error(`Required power: physical GPU count differs from required matrix for ${key}`); + const expectedCount = positive(topology.num_gpus, 'topology num_gpus'); + if (!Number.isSafeInteger(expectedCount)) + throw new Error('Required power: invalid physical GPU count'); + const window = object(declared.measurement_window, 'measurement window'); + const start = positive(window.start_time_unix, 'window start'); + const end = positive(window.end_time_unix, 'window end'); + if (end <= start) throw new Error('Required power: invalid measurement window boundaries'); + if (!Array.isArray(declared.artifacts) || declared.artifacts.length === 0) + throw new Error('Required power: missing required artifacts'); + const evidence = new Map(); + for (const value of declared.artifacts) { + const artifact = object(value, 'required artifact'); + const relative = safeArtifactPath(artifact.path); + if (evidence.has(relative)) + throw new Error(`Required power: duplicate required artifact ${relative}`); + const file = files.get(relative); + if (!file || file.contents === undefined || file.contents.length === 0) + throw new Error(`Required power: missing required artifact ${relative}`); + if ( + artifact.validation_state !== 'valid' || + typeof artifact.sha256 !== 'string' || + !/^[a-f0-9]{64}$/u.test(artifact.sha256) || + createHash('sha256').update(file.contents).digest('hex') !== artifact.sha256 + ) + throw new Error(`Required power: invalid artifact validation or hash ${relative}`); + evidence.set(relative, file); + } + if ( + ![...evidence.values()].some((file) => + file.rows.some((candidate) => isDeepStrictEqual(candidate, row)), + ) + ) + throw new Error(`Required power: benchmark artifact is not hash-bound for ${key}`); + const byName = (name: string): BenchmarkArtifactRows => { + const matches = [...evidence.values()].filter( + (file) => path.posix.basename(file.path) === name, + ); + if (matches.length !== 1) + throw new Error(`Required power: expected one required ${name} artifact`); + return matches[0]; + }; + const sidecars = [...evidence.values()].filter((file) => + /^power_validation.*\.json$/u.test(path.posix.basename(file.path)), + ); + if (sidecars.length !== 1) + throw new Error('Required power: expected one required power validation artifact'); + const audit = object(JSON.parse(sidecars[0].contents!.toString()), 'power validation'); + const auditWindow = object(audit.benchmark_window, 'audit benchmark window'); + if ( + audit.power_valid !== true || + auditWindow.start_time_unix !== start || + auditWindow.end_time_unix !== end || + audit.expected_gpu_count !== expectedCount || + audit.observed_gpu_count !== expectedCount + ) + throw new Error( + `Required power: invalid audit verdict, window or physical GPU coverage for ${key}`, + ); + if (topology.is_multinode && audit.telemetry_kind === 'native_multinode_smi') { + if (!Array.isArray(audit.nodes) || audit.nodes.length === 0) + throw new Error('Required power: missing native node receipts'); + const traces = [...evidence.values()].filter( + (file) => path.posix.basename(file.path) === 'gpu_metrics.csv', + ); + if (traces.length !== audit.nodes.length) + throw new Error('Required power: missing native node telemetry'); + for (const file of traces) { + const directory = path.posix.dirname(file.path); + const manifestFile = evidence.get(`${directory}/manifest.json`); + if (!manifestFile) throw new Error('Required power: missing native node manifest'); + const nodeManifest = object( + JSON.parse(manifestFile.contents!.toString()), + 'native node manifest', + ); + if (nodeManifest.lifecycle !== 'complete' || nodeManifest.collector_exit_code !== 0) + throw new Error('Required power: invalid native node collection'); + const receipt = (audit.nodes as JsonRow[]).find((node) => node.node === nodeManifest.node); + if ( + !receipt || + receipt.manifest_sha256 !== digest(manifestFile) || + receipt.telemetry_sha256 !== digest(file) + ) + throw new Error('Required power: native node receipt differs from evidence'); + for (const hash of [receipt.identity_sha256, receipt.identity_end_sha256]) + if ( + ![...evidence.values()].some( + (item) => path.posix.dirname(item.path) === directory && digest(item) === hash, + ) + ) + throw new Error('Required power: missing native physical identity evidence'); + } + } else if (topology.is_multinode) { + byName('samples.csv'); + byName('manifest.json'); + if ( + ![...evidence.keys()].some((file) => file.includes('/windows/') && file.endsWith('.json')) + ) + throw new Error('Required power: missing central measurement window artifact'); + } else byName('gpu_metrics.csv'); + const energies = object(audit.per_gpu_energy_j, 'audit device energy'); + let auditDevices: JsonRow[]; + if (audit.per_gpu_role === undefined) { + const nodeFiles = [...evidence.values()].filter((file) => + ['power_node.txt', 'gpu_metrics_node.txt'].includes(path.posix.basename(file.path)), + ); + if (nodeFiles.length !== 1) + throw new Error('Required power: expected one node identity artifact'); + const node = nodeFiles[0].contents!.toString().trim(); + const identityFiles = [...evidence.values()].filter((file) => + /^(?:gpu_metrics_identity\.csv|gpu_metrics_devices\.json)$/u.test( + path.posix.basename(file.path), + ), + ); + if (identityFiles.length !== 1) + throw new Error('Required power: expected one physical GPU identity artifact'); + const uuidRows: [string, string][] = []; + if (identityFiles[0].path.endsWith('.csv')) { + const lines = identityFiles[0].contents!.toString().trim().split(/\r?\n/u); + const columns = lines + .shift()! + .split(',') + .map((part) => part.trim().replaceAll('"', '').toLowerCase()); + const indexColumn = columns.indexOf('index'); + const uuidColumn = columns.indexOf('uuid'); + if (indexColumn === -1 || uuidColumn === -1) + throw new Error('Required power: invalid physical GPU identity CSV'); + for (const line of lines) { + const fields = line.split(',').map((part) => part.trim().replaceAll('"', '')); + uuidRows.push([fields[indexColumn], fields[uuidColumn]]); + } + } else { + const visit = (value: unknown): void => { + if (Array.isArray(value)) value.forEach(visit); + else if (value && typeof value === 'object') { + const deviceRow = Object.fromEntries( + Object.entries(value).map(([field, item]) => [field.toLowerCase(), item]), + ); + if ('gpu' in deviceRow && 'uuid' in deviceRow) + uuidRows.push([String(deviceRow.gpu), String(deviceRow.uuid)]); + else Object.values(value).forEach(visit); + } + }; + visit(JSON.parse(identityFiles[0].contents!.toString())); + } + if ( + uuidRows.length === 0 || + uuidRows.some( + ([index, uuid]) => + !/^\d+$/u.test(index) || !uuid || ['n/a', 'none', 'null'].includes(uuid.toLowerCase()), + ) || + new Set(uuidRows.map(([index]) => index)).size !== uuidRows.length || + new Set(uuidRows.map(([, uuid]) => uuid)).size !== uuidRows.length + ) + throw new Error('Required power: invalid or duplicate physical GPU identity'); + const uuids = new Map(uuidRows); + auditDevices = Object.entries(energies).map(([index, energy]) => ({ + node, + gpu_uuid: uuids.get(index), + role: 'aggregate', + energy_j: energy, + })); + } else { + const roles = object(audit.per_gpu_role, 'audit device roles'); + const nativeNodes = new Map(); + if (audit.telemetry_kind === 'native_multinode_smi') + for (const value of audit.nodes as unknown[]) { + const receipt = object(value, 'native node receipt'); + for (const uuid of Object.values( + object(receipt.physical_gpu_ids, 'native GPU identities'), + )) { + const uuidValue = nonempty(uuid, 'native GPU UUID'); + if (nativeNodes.has(uuidValue)) + throw new Error('Required power: duplicate native GPU identity'); + nativeNodes.set(uuidValue, nonempty(receipt.node, 'native node')); + } + } + auditDevices = Object.entries(energies).map(([device, energy]) => { + const slash = device.lastIndexOf('/'); + return { + node: nativeNodes.get(device) ?? device.slice(0, slash), + gpu_uuid: nativeNodes.has(device) ? device : device.slice(slash + 1), + role: roles[device] === 'agg' ? 'aggregate' : roles[device], + energy_j: energy, + }; + }); + } + if ( + !Array.isArray(declared.devices) || + declared.devices.length !== expectedCount || + auditDevices.length !== expectedCount + ) + throw new Error(`Required power: missing physical GPU evidence for ${key}`); + const devices = new Set(); + const roles = new Map(); + let totalEnergy = 0; + const roleEnergy = new Map(); + const nodes = new Set(); + for (const value of declared.devices) { + const device = object(value, 'physical GPU'); + const node = nonempty(device.node, 'GPU node'); + nodes.add(node); + const uuid = nonempty(device.gpu_uuid, 'physical GPU UUID'); + if (['n/a', 'none', 'null'].includes(uuid.toLowerCase()) || devices.has(uuid)) + throw new Error('Required power: duplicate or invalid physical GPU UUID'); + devices.add(uuid); + if (!['aggregate', 'prefill', 'decode'].includes(String(device.role))) + throw new Error('Required power: invalid physical GPU role'); + roles.set(String(device.role), (roles.get(String(device.role)) ?? 0) + 1); + const energy = positive(device.energy_j, 'device energy_j'); + totalEnergy += energy; + roleEnergy.set(String(device.role), (roleEnergy.get(String(device.role)) ?? 0) + energy); + if (!auditDevices.some((actual) => isDeepStrictEqual(actual, device))) + throw new Error(`Required power: physical GPU evidence differs from audit for ${key}`); + } + if (nodes.size !== (matrix['node-count'] ?? 1)) + throw new Error( + `Required power: participating node count differs from required matrix for ${key}`, + ); + if (topology.disagg) { + if ( + roles.get('prefill') !== positive(topology.num_prefill_gpu, 'prefill GPU count') || + roles.get('decode') !== positive(topology.num_decode_gpu, 'decode GPU count') || + roles.has('aggregate') + ) + throw new Error(`Required power: missing prefill or decode evidence for ${key}`); + for (const role of ['prefill', 'decode']) { + const energy = positive(row[`${role}_gpu_energy_j`], `${role} energy`); + if (Math.abs((roleEnergy.get(role) ?? 0) - energy) > Math.max(0.01, energy * 1e-4)) + throw new Error(`Required power: ${role} energy differs from device evidence for ${key}`); + } + } else if (roles.get('aggregate') !== expectedCount) { + throw new Error(`Required power: invalid aggregate GPU roles for ${key}`); + } + const energy = positive(row.total_gpu_energy_j, 'total_gpu_energy_j'); + if (Math.abs(totalEnergy - energy) > Math.max(0.01, energy * 1e-4)) + throw new Error(`Required power: total GPU energy differs from device evidence for ${key}`); + } +} diff --git a/packages/db/src/etl/skip-tracker.test.ts b/packages/db/src/etl/skip-tracker.test.ts index e407db3a3..1851b0417 100644 --- a/packages/db/src/etl/skip-tracker.test.ts +++ b/packages/db/src/etl/skip-tracker.test.ts @@ -9,6 +9,7 @@ describe('createSkipTracker', () => { expect(tracker.skips.unmappedHw).toBe(0); expect(tracker.skips.noIslOsl).toBe(0); expect(tracker.skips.dbError).toBe(0); + expect(tracker.skips.telemetryError).toBe(0); expect(tracker.skips.traceReplayMissing).toBe(0); }); diff --git a/packages/db/src/etl/skip-tracker.ts b/packages/db/src/etl/skip-tracker.ts index 5d485bf22..055a92b46 100644 --- a/packages/db/src/etl/skip-tracker.ts +++ b/packages/db/src/etl/skip-tracker.ts @@ -10,6 +10,15 @@ export interface Skips { noIslOsl: number; failedRun: number; dbError: number; + /** + * PowerX telemetry digest failures, counted apart from `dbError` because they + * are not fatal. The benchmark rows land either way; only the per-point + * telemetry tab is affected, and the artifact can be re-digested later by + * `admin:db:backfill-gpu-metrics`. Folding these into `dbError` would let one + * malformed `gpu_metrics_*` CSV turn the whole production ingest red, through + * the publication manifest that verify-power-publication treats as fatal. + */ + telemetryError: number; /** Agentic point whose sibling `agentic_` artifact had no trace_replay files. */ traceReplayMissing: number; } @@ -35,6 +44,15 @@ export interface SkipTracker { * @param err - The caught error. */ recordDbError: (context: string, err: Error) => void; + /** + * Record a non-fatal PowerX telemetry digest failure. Same printing and + * suppression as `recordDbError`, but increments `skips.telemetryError` so the + * ingest is not failed by it. + * + * @param context - Human-readable label for where the error occurred. + * @param err - The caught error. + */ + recordTelemetryError: (context: string, err: Error) => void; /** * Capture a point-in-time snapshot of the current skip counters and * unmapped-name sets. Used together with `diff()` to report per-artifact drops. @@ -76,12 +94,14 @@ export function createSkipTracker(): SkipTracker { noIslOsl: 0, failedRun: 0, dbError: 0, + telemetryError: 0, traceReplayMissing: 0, }; const unmappedModels = new Set(); const unmappedHws = new Set(); const unmappedPrecisions = new Set(); let dbErrorsPrinted = 0; + let telemetryErrorsPrinted = 0; return { skips, @@ -100,6 +120,19 @@ export function createSkipTracker(): SkipTracker { } }, + recordTelemetryError(context: string, err: Error): void { + skips.telemetryError++; + if (telemetryErrorsPrinted < MAX_DB_ERRORS) { + console.error(` [TELEMETRY] ${context}: ${err.message}`); + telemetryErrorsPrinted++; + if (telemetryErrorsPrinted === MAX_DB_ERRORS) { + console.error( + ' [TELEMETRY] further telemetry errors suppressed; count included in summary', + ); + } + } + }, + snapshot(): SkipSnapshot { return { model: skips.unmappedModel, diff --git a/packages/db/src/etl/telemetry-receipt.test.ts b/packages/db/src/etl/telemetry-receipt.test.ts new file mode 100644 index 000000000..1fba454e7 --- /dev/null +++ b/packages/db/src/etl/telemetry-receipt.test.ts @@ -0,0 +1,274 @@ +import { readFileSync } from 'node:fs'; +import { PGlite } from '@electric-sql/pglite'; +import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest'; + +import type { Sql } from './db-utils'; +import { stablePowerPointIdentity } from './power-publication'; + +import { + readTelemetryReceipt, + summarizeTelemetryReceipt, + telemetryArtifactsForAttempt, + verifyTelemetryApi, + type TelemetryReceipt, +} from './telemetry-receipt'; + +describe('telemetry attachment receipts', () => { + it('isolates the source attempt before selecting or downloading artifact pairs', () => { + const old = { + name: 'gpu_metrics_old', + created_at: '2026-09-20T00:00:00Z', + archive_download_url: 'old', + }; + const current = { + name: 'gpu_metrics_new', + created_at: '2026-09-21T01:00:00Z', + archive_download_url: 'new', + }; + const source = { run_attempt: 2, run_started_at: '2026-09-21T00:00:00Z' }; + expect(telemetryArtifactsForAttempt([old, current], source, 2)).toEqual([current]); + expect(() => telemetryArtifactsForAttempt([old, current], source, 1)).toThrow('mixed-attempt'); + expect(() => telemetryArtifactsForAttempt([current], { run_attempt: 2 }, 2)).toThrow( + 'start time', + ); + }); +}); + +const run = { runId: 123, runAttempt: 2 }; +const artifactName = 'gpu_metrics_shared'; + +function receiptFixture(): TelemetryReceipt { + const series = [{ id: 1, artifactName, fileName: 'gpu_metrics.csv', sampleCount: 2 }]; + return { + version: 1, + ...run, + checkedAt: '2026-09-21T00:00:00Z', + expectedSource: 'benchmark_artifacts', + plannedPointCount: null, + counts: {} as TelemetryReceipt['counts'], + points: [10, 11].map((id) => ({ + key: String(id), + identity: { conc: id }, + benchmarkResultId: id, + artifactNames: [artifactName], + produced: true, + series, + storage: { status: 'complete', expectedSeries: series }, + linkedSeriesIds: [1], + api: { status: 'readable' }, + reasons: [], + recovery: { ...run, artifactNames: [artifactName] }, + })), + }; +} + +describe('telemetry completeness accounting', () => { + it('counts shared storage once while retaining both independently linked points', () => { + const { counts } = summarizeTelemetryReceipt(receiptFixture()); + expect(counts).toMatchObject({ + expectedPoints: 2, + producedPoints: 2, + producedArtifacts: 1, + storedPoints: 2, + linkedPoints: 2, + storedSeries: 1, + storedSamples: 2, + apiCompletePoints: 2, + }); + }); + + it.each([ + { expectedSource: 'unknown' as const }, + { + expectationErrors: [ + { benchmarkArtifact: 'bmk_missing', artifactNames: [], error: 'unreadable' }, + ], + }, + ])('keeps an unknown expectation denominator with %j', (unknown) => { + const result = summarizeTelemetryReceipt({ ...receiptFixture(), ...unknown }); + expect(result.counts).toMatchObject({ + expectedPoints: null, + storedPoints: 2, + storedSamples: 2, + }); + }); + + it.each([ + { databaseError: 'database unavailable' }, + { recoveryError: 'corrective ingest failed' }, + ])('does not promote readable old data to complete while %j', (failure) => { + const result = summarizeTelemetryReceipt({ ...receiptFixture(), ...failure }); + expect(result.counts.apiReadablePoints).toBe(2); + expect(result.counts.apiCompletePoints).toBe(0); + expect(result.counts.storedSamples).toBe('databaseError' in failure ? null : 2); + }); + + it('does not count unlinked or failed corrected points as API-complete', () => { + const receipt = receiptFixture(); + receipt.points[0]!.linkedSeriesIds = []; + receipt.points[1]!.error = 'correction failed'; + expect(summarizeTelemetryReceipt(receipt).counts).toMatchObject({ + storedPoints: 2, + linkedPoints: 1, + apiReadablePoints: 2, + apiCompletePoints: 0, + }); + }); +}); + +describe('telemetry API verification', () => { + it.each([ + ['matching data', 200, 10, 2, 2, true], + ['missing endpoint', 404, 10, 2, 2, false], + ['unavailable endpoint', 503, 10, 2, 2, false], + ['different point', 200, 11, 2, 2, false], + ['different sample count', 200, 10, 3, 2, false], + ['truncated data', 200, 10, 2, 1, false], + ] as const)('%s', async (_name, status, id, sampleCount, dataCount, readable) => { + const receipt = receiptFixture(); + receipt.points = [receipt.points[0]!]; + receipt.points[0]!.api = { status: 'unknown' }; + const fetcher = vi.fn().mockResolvedValue( + Response.json( + { + benchmarkResultId: id, + series: [ + { id: 1, sampleCount, data: Array.from({ length: dataCount }, () => ({ power: 100 })) }, + ], + }, + { status }, + ), + ); + const result = await verifyTelemetryApi(receipt, 'https://example.test', { fetch: fetcher }); + expect(String(fetcher.mock.calls[0]![0])).toBe( + 'https://example.test/api/v1/gpu-metrics-point?id=10', + ); + expect(result.points[0]!.api.status).toBe(readable ? 'readable' : 'failed'); + expect(result.counts.apiCompletePoints).toBe(readable ? 1 : 0); + if (!readable) expect(result.points[0]!.api.error).toBeTruthy(); + }); + + it('requires the complete stored series set and skips points without database identities', async () => { + const receipt = receiptFixture(); + receipt.points[1]!.benchmarkResultId = null; + receipt.points[1]!.api = { status: 'unknown' }; + const fetcher = vi + .fn() + .mockResolvedValue(Response.json({ benchmarkResultId: 10, series: [] })); + const result = await verifyTelemetryApi(receipt, 'https://example.test', { fetch: fetcher }); + expect(fetcher).toHaveBeenCalledTimes(1); + expect(result.points[0]!.api.error).toContain('series set differs'); + expect(result.points[1]!.api.status).toBe('unknown'); + expect(result.counts.apiCompletePoints).toBe(0); + }); +}); + +describe('persisted telemetry receipts', () => { + let db: PGlite; + let sql: Sql; + beforeAll(async () => { + db = await PGlite.create(); + for (const name of ['001_initial_schema.sql', '016_gpu_metrics.sql']) { + await db.exec(readFileSync(new URL(`../../migrations/${name}`, import.meta.url), 'utf8')); + } + await db.exec( + "ALTER TABLE benchmark_results ADD COLUMN offload_mode text DEFAULT 'off'; ALTER TABLE benchmark_results ADD COLUMN recipe_fingerprint text", + ); + sql = (async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await db.query(query, values); + return result.rows; + }) as unknown as Sql; + }, 20_000); + afterAll(async () => { + await db?.close(); + }); + beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, created_at, date) + VALUES (1, 123, 2, 'Sweep', '2026-09-21', '2026-09-21'); + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', 1, 1, 1, 1); + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, isl, osl, conc, metrics) + VALUES (10, 1, 1, 'single_turn', '2026-09-21', 8192, 1024, 1, '{}'), + (11, 1, 1, 'single_turn', '2026-09-21', 8192, 1024, 2, '{}'); + INSERT INTO gpu_metric_series (id, workflow_run_id, artifact_name, config_key, file_name, vendor, csv_sha256, sample_count, gpu_count, started_at, ended_at) + VALUES (1, 1, 'gpu_metrics_shared', 'shared', 'gpu_metrics.csv', 'nvidia', 'source', 2, 1, '2026-09-21T00:00:00Z', '2026-09-21T00:00:01Z'); + INSERT INTO gpu_metric_samples (series_id, gpu_index, sampled_at, power_w) + VALUES (1, 0, '2026-09-21T00:00:00Z', 100), (1, 0, '2026-09-21T00:00:01Z', 200); + INSERT INTO benchmark_result_gpu_metrics VALUES (10, 1), (11, 1);`); + }); + + async function inventory(value: unknown) { + await db.query('UPDATE gpu_metric_series SET sidecars = $1::jsonb', [JSON.stringify(value)]); + } + const completeInventory = { seriesInventory: [{ fileName: 'gpu_metrics.csv', sampleCount: 2 }] }; + + it.each([false, true])( + 'reads complete JSONB inventory (encoded=%s) and preserves unrelated targeted state', + async (encoded) => { + await inventory(encoded ? JSON.stringify(completeInventory) : completeInventory); + const original = await readTelemetryReceipt(sql, run, []); + expect(original.counts).toMatchObject({ + expectedPoints: 2, + storedPoints: 2, + linkedPoints: 2, + storedSeries: 1, + storedSamples: 2, + apiCompletePoints: 0, + }); + original.points[1]!.api = { status: 'readable' }; + const first = original.points[0]!; + const repaired = await readTelemetryReceipt( + sql, + run, + [{ identity: first.identity, artifactNames: [artifactName], produced: true }], + { previous: original, targeted: true }, + ); + expect( + repaired.points.find((point) => point.key === stablePowerPointIdentity(first.identity))!.api + .status, + ).toBe('unknown'); + expect(repaired.points[1]).toEqual(original.points[1]); + }, + ); + + it('keeps missing inventory unknown and missing host storage incomplete', async () => { + await inventory('{invalid json'); + const unknown = await readTelemetryReceipt(sql, run, []); + expect(unknown.databaseError).toBeUndefined(); + expect(unknown.counts).toMatchObject({ + storedPoints: 0, + storageUnknownPoints: 2, + storedSamples: 2, + }); + await inventory({ + seriesInventory: [ + ...completeInventory.seriesInventory, + { fileName: 'host-b/gpu_metrics.csv', sampleCount: 2 }, + ], + }); + const incomplete = await readTelemetryReceipt(sql, run, []); + expect(incomplete.points[0]!.storage.status).toBe('incomplete'); + expect(incomplete.points[0]!.reasons).toContain('series_coverage_incomplete'); + expect(incomplete.counts.storedPoints).toBe(0); + }); + + it('reports failed SQL as unknown storage rather than a successful empty receipt', async () => { + await db.exec('ALTER TABLE gpu_metric_series RENAME TO unavailable_gpu_metric_series'); + try { + const result = await readTelemetryReceipt(sql, run, []); + expect(result.databaseError).toContain('gpu_metric_series'); + expect(result.counts).toMatchObject({ + expectedPoints: null, + storedPoints: null, + linkedPoints: null, + storedSeries: null, + storedSamples: null, + apiCompletePoints: 0, + }); + } finally { + await db.exec('ALTER TABLE unavailable_gpu_metric_series RENAME TO gpu_metric_series'); + } + }); +}); diff --git a/packages/db/src/etl/telemetry-receipt.ts b/packages/db/src/etl/telemetry-receipt.ts new file mode 100644 index 000000000..16b8b019b --- /dev/null +++ b/packages/db/src/etl/telemetry-receipt.ts @@ -0,0 +1,461 @@ +import type { Sql } from './db-utils'; +import type { ArtifactMeta, RunMeta } from '../lib/github-artifacts'; +import { stablePowerPointIdentity } from './power-publication'; +import { gpuMetricsArtifactSuffix } from './gpu-metrics-artifacts'; + +export interface TelemetryObservation { + identity: Record; + /** Exact selected artifact names, or both allowed sibling names when absent. */ + artifactNames: string[]; + produced: boolean | null; + error?: string; +} + +interface TelemetrySeriesCoverage { + artifactName: string; + fileName: string; + sampleCount: number; +} + +export interface TelemetryPointReceipt extends TelemetryObservation { + key: string; + benchmarkResultId: number | null; + series: (TelemetrySeriesCoverage & { id: number })[]; + storage: { + status: 'complete' | 'incomplete' | 'unknown'; + /** Null means the original artifact inventory was not retained. */ + expectedSeries: TelemetrySeriesCoverage[] | null; + }; + linkedSeriesIds: number[]; + api: { + status: 'unknown' | 'readable' | 'failed'; + checkedAt?: string; + url?: string; + error?: string; + }; + reasons: string[]; + recovery: { runId: number; runAttempt: number; artifactNames: string[] }; +} + +export interface TelemetryReceipt { + version: 1; + runId: number; + runAttempt: number; + checkedAt: string; + /** This is attachment coverage for known benchmark points, not a full-sweep verdict. */ + expectedSource: 'benchmark_artifacts' | 'database_benchmarks' | 'unknown'; + /** No complete independent planned matrix is available to this attachment reader. */ + plannedPointCount: null; + databaseError?: string; + recoveryError?: string; + /** Missing on older receipts or when the failed recovery covered the whole run. */ + recoveryArtifactNames?: string[]; + /** The sibling is known, but a download/parse failure prevents identifying its points. */ + expectationErrors?: { benchmarkArtifact: string; artifactNames: string[]; error: string }[]; + points: TelemetryPointReceipt[]; + counts: { + expectedPoints: number | null; + producedPoints: number; + productionUnknownPoints: number; + storedPoints: number | null; + linkedPoints: number | null; + storageUnknownPoints: number; + /** Actual successful HTTP reads, including readable but incomplete artifacts. */ + apiReadablePoints: number; + /** Readable, fully stored/linked points with no outstanding corrective-ingest failure. */ + apiCompletePoints: number; + apiUnknownPoints: number; + producedArtifacts: number; + storedSeries: number | null; + storedSamples: number | null; + }; +} + +interface StoredPoint extends Record { + id: number; +} + +interface StoredSeries { + id: number; + artifact_name: string; + file_name: string; + sidecars: unknown; + sample_count: number; + benchmark_result_ids: number[]; +} + +function isInventoryEntry(value: unknown): value is { fileName: string; sampleCount: number } { + if (value === null || typeof value !== 'object') return false; + const entry = value as Record; + return ( + typeof entry.fileName === 'string' && + typeof entry.sampleCount === 'number' && + Number.isSafeInteger(entry.sampleCount) && + entry.sampleCount >= 0 + ); +} + +/** JSONB may contain an encoded object; unavailable inventories remain unknown. */ +function storedSeriesInventory(sidecars: unknown): unknown { + if (typeof sidecars === 'string') { + try { + sidecars = JSON.parse(sidecars) as unknown; + } catch { + return undefined; + } + } + return sidecars !== null && typeof sidecars === 'object' && !Array.isArray(sidecars) + ? (sidecars as { seriesInventory?: unknown }).seriesInventory + : undefined; +} + +function seriesCoverageKey(artifact: string, file: string): string { + return JSON.stringify([artifact, file]); +} + +function seriesCoverage( + series: readonly StoredSeries[], + reasons: string[], +): TelemetryPointReceipt['storage'] { + if (series.length === 0) return { status: 'incomplete', expectedSeries: null }; + const expected = new Map(); + const inventories = new Map(); + let unknown = false; + for (const entry of series) { + const inventory = storedSeriesInventory(entry.sidecars); + if (!Array.isArray(inventory) || inventory.length === 0 || !inventory.every(isInventoryEntry)) { + unknown = true; + continue; + } + const signature = JSON.stringify( + inventory.toSorted((a, b) => a.fileName.localeCompare(b.fileName)), + ); + const previous = inventories.get(entry.artifact_name); + if (previous && previous !== signature) unknown = true; + inventories.set(entry.artifact_name, signature); + if (!inventory.some((item) => item.fileName === entry.file_name)) unknown = true; + for (const item of inventory) { + expected.set(seriesCoverageKey(entry.artifact_name, item.fileName), { + artifactName: entry.artifact_name, + ...item, + }); + } + } + const actual = new Map( + series.map((entry) => [seriesCoverageKey(entry.artifact_name, entry.file_name), entry]), + ); + const missing = [...expected].some(([name]) => !actual.has(name)); + const wrongSamples = [...expected].some(([name, entry]) => { + const stored = actual.get(name); + return stored !== undefined && Number(stored.sample_count) !== entry.sampleCount; + }); + if (missing) reasons.push('series_coverage_incomplete'); + if (wrongSamples) reasons.push('sample_coverage_incomplete'); + if (unknown) reasons.push('series_inventory_unknown'); + return { + status: + missing || wrongSamples || series.some((entry) => Number(entry.sample_count) === 0) + ? 'incomplete' + : unknown + ? 'unknown' + : 'complete', + expectedSeries: expected.size > 0 ? [...expected.values()] : null, + }; +} + +/** GitHub's run-wide listing also retains artifacts from earlier attempts. */ +export function telemetryArtifactsForAttempt( + artifacts: readonly ArtifactMeta[], + source: Pick, + targetAttempt: number, +): ArtifactMeta[] { + if (source.run_attempt !== targetAttempt) + throw new Error( + `GitHub attempt ${source.run_attempt} differs from target ${targetAttempt}; refusing mixed-attempt telemetry recovery`, + ); + const start = Date.parse(source.run_started_at ?? ''); + if (!Number.isFinite(start)) + throw new Error('Cannot isolate telemetry artifacts without attempt start time'); + return artifacts.filter((artifact) => Date.parse(artifact.created_at) >= start); +} + +/** Count identities, artifacts, series and samples separately; shared series count once. */ +export function summarizeTelemetryReceipt(receipt: TelemetryReceipt): TelemetryReceipt { + const points = receipt.points; + const series = new Map(points.flatMap((point) => point.series.map((s) => [s.id, s] as const))); + const complete = points.filter((point) => point.storage?.status === 'complete'); + const linked = complete.filter((point) => + point.series.every((s) => point.linkedSeriesIds.includes(s.id)), + ); + receipt.counts = { + expectedPoints: + receipt.expectedSource === 'unknown' || receipt.expectationErrors?.length + ? null + : points.length, + producedPoints: points.filter((p) => p.produced === true).length, + productionUnknownPoints: points.filter((p) => p.produced === null).length, + storedPoints: receipt.databaseError ? null : complete.length, + linkedPoints: receipt.databaseError ? null : linked.length, + storageUnknownPoints: points.filter( + (point) => !point.storage || point.storage.status === 'unknown', + ).length, + apiReadablePoints: points.filter((p) => p.api.status === 'readable').length, + apiCompletePoints: + receipt.databaseError || receipt.recoveryError + ? 0 + : linked.filter((point) => point.api.status === 'readable' && !point.error).length, + apiUnknownPoints: points.filter((p) => p.api.status === 'unknown').length, + producedArtifacts: new Set( + points.filter((p) => p.produced === true).flatMap((p) => p.artifactNames), + ).size, + storedSeries: receipt.databaseError ? null : series.size, + storedSamples: receipt.databaseError + ? null + : [...series.values()].reduce((sum, s) => sum + s.sampleCount, 0), + }; + return receipt; +} + +/** Absent observations name alternative siblings, not two required uploads. */ +function coversArtifactScope( + current: TelemetryObservation, + previous: TelemetryObservation, +): boolean { + return ( + previous.artifactNames.length > 0 && + previous.artifactNames.every( + (name) => + current.artifactNames.includes(name) || + (previous.produced === false && + gpuMetricsArtifactSuffix(name) !== null && + current.artifactNames.some( + (candidate) => gpuMetricsArtifactSuffix(candidate) === gpuMetricsArtifactSuffix(name), + )), + ) + ); +} + +/** Read actual persistence/link state, preserving unrelated points on targeted recovery. */ +export async function readTelemetryReceipt( + sql: Sql, + run: { runId: number; runAttempt: number }, + observations: readonly TelemetryObservation[], + options: { + expectedSource?: TelemetryReceipt['expectedSource']; + previous?: TelemetryReceipt; + targeted?: boolean; + /** Mapped identity -> persisted ID, only from the historical resolver's unique fallback. */ + uniqueFallbacks?: ReadonlyMap; + } = {}, +): Promise { + const previous = options.previous; + if (previous && (previous.runId !== run.runId || previous.runAttempt !== run.runAttempt)) + throw new Error('Telemetry receipt run/attempt does not match the recovery target'); + const receipt: TelemetryReceipt = { + version: 1, + ...run, + checkedAt: new Date().toISOString(), + expectedSource: options.expectedSource ?? previous?.expectedSource ?? 'database_benchmarks', + plannedPointCount: null, + points: [], + counts: {} as TelemetryReceipt['counts'], + }; + let rows: StoredPoint[] = []; + let series: StoredSeries[] = []; + try { + rows = await sql` + select c.*, br.id, br.benchmark_type, br.isl, br.osl, br.conc, + br.offload_mode, br.recipe_fingerprint + from benchmark_results br + join configs c on c.id = br.config_id + join workflow_runs wr on wr.id = br.workflow_run_id + where wr.github_run_id = ${run.runId} and wr.run_attempt = ${run.runAttempt} + order by br.id + `; + series = await sql` + select s.id, s.artifact_name, s.file_name, s.sidecars, + (select count(*)::int from gpu_metric_samples x where x.series_id = s.id) as sample_count, + coalesce((select array_agg(l.benchmark_result_id order by l.benchmark_result_id) + from benchmark_result_gpu_metrics l + join benchmark_results br on br.id = l.benchmark_result_id + where l.series_id = s.id and br.workflow_run_id = s.workflow_run_id), '{}') as benchmark_result_ids + from gpu_metric_series s join workflow_runs wr on wr.id = s.workflow_run_id + where wr.github_run_id = ${run.runId} and wr.run_attempt = ${run.runAttempt} + order by s.id + `; + } catch (error) { + receipt.databaseError = error instanceof Error ? error.message : String(error); + if (observations.length === 0 && !previous) receipt.expectedSource = 'unknown'; + } + const stored = new Map(rows.map((row) => [stablePowerPointIdentity(row), row])); + const storedById = new Map(rows.map((row) => [Number(row.id), row])); + const prior = new Map(previous?.points.map((point) => [point.key, point])); + const observed = new Map( + observations.map((point) => [stablePowerPointIdentity(point.identity), point]), + ); + for (const [key, id] of options.uniqueFallbacks ?? []) { + const observation = observed.get(key); + const row = storedById.get(id); + if (receipt.databaseError || !observation || !row || stored.has(key)) continue; + // The resolver allows historical offload drift only; keep every other dimension exact. + const identity = { ...observation.identity, offload_mode: row.offload_mode }; + const canonicalKey = stablePowerPointIdentity(identity); + if (canonicalKey !== stablePowerPointIdentity(row)) continue; + const collision = observed.get(canonicalKey); + const priorCanonical = prior.get(canonicalKey); + if ( + (priorCanonical && + (priorCanonical.artifactNames.length > 0 || priorCanonical.error) && + !coversArtifactScope(observation, priorCanonical)) || + (collision && + (collision.produced !== observation.produced || + collision.error !== observation.error || + collision.artifactNames.length !== observation.artifactNames.length || + collision.artifactNames.some((name) => !observation.artifactNames.includes(name)))) + ) { + receipt.expectedSource = 'unknown'; + continue; + } + observed.delete(key); + observed.set(canonicalKey, { ...observation, identity }); + const old = prior.get(key); + if (old?.benchmarkResultId === null && coversArtifactScope(observation, old)) prior.delete(key); + } + const resolvedObservations = [...observed.values()]; + const pendingRecoveryArtifacts = previous?.recoveryArtifactNames?.filter( + (name) => + !resolvedObservations.some( + (point) => point.produced === true && !point.error && point.artifactNames.includes(name), + ), + ); + const successfulKeys = new Set( + resolvedObservations + .filter((point) => point.produced === true && !point.error) + .map((point) => stablePowerPointIdentity(point.identity)), + ); + const recoveryUnresolved = previous?.recoveryArtifactNames?.length + ? pendingRecoveryArtifacts?.length + : options.targeted || + resolvedObservations.length === 0 || + successfulKeys.size !== resolvedObservations.length || + [...prior.values()].some((point) => !successfulKeys.has(point.key)); + Object.assign(receipt, { + ...(previous?.recoveryError && recoveryUnresolved + ? { + recoveryError: previous.recoveryError, + ...(pendingRecoveryArtifacts ? { recoveryArtifactNames: pendingRecoveryArtifacts } : {}), + } + : {}), + ...(previous?.expectationErrors + ? { + expectationErrors: previous.expectationErrors.filter( + (error) => + !resolvedObservations.some((point) => + point.artifactNames.some((name) => error.artifactNames.includes(name)), + ), + ), + } + : {}), + }); + const keys = new Set([...prior.keys(), ...stored.keys(), ...observed.keys()]); + for (const key of keys) { + const old = prior.get(key); + if (options.targeted && old && !observed.has(key)) { + receipt.points.push(old); + continue; + } + const row = stored.get(key); + const benchmarkResultId = row ? Number(row.id) : null; + const linked = series.filter((s) => + s.benchmark_result_ids.map(Number).includes(benchmarkResultId!), + ); + const observation = observed.get(key) ?? + old ?? { + identity: row!, + artifactNames: [...new Set(linked.map((s) => s.artifact_name))], + produced: linked.length > 0 ? true : null, + }; + const matching = series.filter((s) => observation.artifactNames.includes(s.artifact_name)); + const reasons: string[] = []; + if (observation.produced === false) reasons.push('artifact_missing'); + if (observation.produced === null) reasons.push('artifact_pair_unknown'); + if (observation.error) reasons.push('ingest_failed'); + if (receipt.databaseError) reasons.push('database_check_failed'); + else { + if (benchmarkResultId === null) reasons.push('benchmark_not_stored'); + if (matching.length === 0) reasons.push('series_not_stored'); + else if (matching.some((s) => Number(s.sample_count) === 0)) reasons.push('samples_missing'); + if (matching.some((s) => !linked.includes(s))) reasons.push('point_link_missing'); + } + const storage: TelemetryPointReceipt['storage'] = receipt.databaseError + ? { status: 'unknown', expectedSeries: null } + : seriesCoverage(matching, reasons); + receipt.points.push({ + identity: observation.identity, + artifactNames: observation.artifactNames, + produced: observation.produced, + ...(observation.error ? { error: observation.error } : {}), + key, + benchmarkResultId, + series: matching.map((s) => ({ + id: Number(s.id), + artifactName: s.artifact_name, + fileName: s.file_name, + sampleCount: Number(s.sample_count), + })), + storage, + linkedSeriesIds: linked.filter((s) => matching.includes(s)).map((s) => Number(s.id)), + api: { status: 'unknown' }, + reasons, + recovery: { ...run, artifactNames: observation.artifactNames }, + }); + } + receipt.points.sort((a, b) => a.key.localeCompare(b.key)); + return summarizeTelemetryReceipt(receipt); +} + +/** Only an actual HTTP read advances API status; DB-queryable is not API-readable. */ +export async function verifyTelemetryApi( + receipt: TelemetryReceipt, + origin: string, + options: { fetch?: typeof fetch; headers?: HeadersInit } = {}, +): Promise { + for (const point of receipt.points) { + if (point.benchmarkResultId === null) continue; + const url = new URL('/api/v1/gpu-metrics-point', origin); + url.searchParams.set('id', String(point.benchmarkResultId)); + const checkedAt = new Date().toISOString(); + try { + const response = await (options.fetch ?? fetch)(url, { + headers: options.headers, + signal: AbortSignal.timeout(30_000), + }); + if (!response.ok) throw new Error(`HTTP ${response.status}`); + const payload = await response.json(); + if (payload?.benchmarkResultId !== point.benchmarkResultId || !Array.isArray(payload.series)) + throw new Error('Point API returned a different identity or invalid series'); + if (point.series.length === 0 || payload.series.length !== point.series.length) + throw new Error('Point API series set differs from stored telemetry'); + for (const stored of point.series) { + const actual = payload.series.find((s: { id: number }) => s.id === stored.id); + if ( + !actual || + stored.sampleCount === 0 || + actual.sampleCount !== stored.sampleCount || + !Array.isArray(actual.data) || + actual.data.length !== stored.sampleCount + ) + throw new Error(`Point API series ${stored.id} sample count differs from storage`); + } + point.api = { status: 'readable', checkedAt, url: url.toString() }; + } catch (error) { + point.api = { + status: 'failed', + checkedAt, + url: url.toString(), + error: error instanceof Error ? error.message : String(error), + }; + } + } + return summarizeTelemetryReceipt(receipt); +} diff --git a/packages/db/src/ingest-ci-run.test.ts b/packages/db/src/ingest-ci-run.test.ts index d3ad330b4..520028343 100644 --- a/packages/db/src/ingest-ci-run.test.ts +++ b/packages/db/src/ingest-ci-run.test.ts @@ -5,6 +5,47 @@ import path from 'node:path'; import { fileURLToPath } from 'node:url'; import { describe, expect, it } from 'vitest'; +describe('explicit download attempt identity', () => { + it('rejects a download whose GitHub attempt differs from the requested one', () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'attempt-ingest-')); + const manifestPath = path.join(dir, 'power-publication.json'); + try { + fs.writeFileSync( + path.join(dir, 'gh'), + '#!/bin/sh\ncase "$*" in\n*--paginate*) exit 0 ;;\n*--jq*) printf "2\\n" ;;\n*) exit 3 ;;\nesac\n', + { mode: 0o755 }, + ); + const result = spawnSync( + 'bun', + [ + fileURLToPath(new URL('ingest-ci-run.ts', import.meta.url)), + '--download', + 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/25199291771/attempts/1', + ], + { + cwd: dir, + env: { + PATH: `${dir}${path.delimiter}${process.env.PATH}`, + TMPDIR: dir, + NODE_ENV: 'test', + DATABASE_WRITE_URL: 'postgres://unused:unused@127.0.0.1:1/unused', + GITHUB_TOKEN: 'unused', + POWER_PUBLICATION_MANIFEST: manifestPath, + }, + encoding: 'utf8', + timeout: 10_000, + }, + ); + expect(result.error).toBeUndefined(); + expect(result.status, result.stderr).toBe(1); + expect(result.stderr).toContain('GitHub attempt 2 differs from requested 1'); + expect(fs.existsSync(manifestPath)).toBe(false); + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } + }); +}); + describe('purged CI ingestion', () => { it.each([ { runId: 20286769842, runAttempt: 1, reused: false }, @@ -53,6 +94,7 @@ describe('purged CI ingestion', () => { runAttempt, points: [], ingestErrors: [], + telemetryWarnings: [], }); } finally { fs.rmSync(dir, { recursive: true, force: true }); diff --git a/packages/db/src/ingest-ci-run.ts b/packages/db/src/ingest-ci-run.ts index a27225bc9..351233bf9 100644 --- a/packages/db/src/ingest-ci-run.ts +++ b/packages/db/src/ingest-ci-run.ts @@ -27,6 +27,8 @@ import { createHash } from 'node:crypto'; import { powerPublicationPoint, publicationIdentity, + benchmarkPublicationIdentity, + stablePowerPointIdentity, type PowerPublicationPoint, } from './etl/power-publication'; import os from 'os'; @@ -60,6 +62,7 @@ import { readReusedIngestMetadata, } from './etl/reused-ingest-metadata'; import { mapBenchmarkRow, type BenchmarkParams } from './etl/benchmark-mapper'; +import { preflightRequiredPowerCurves } from './etl/required-power-curve'; import { assertRequiredPowerPointsRetained, verifyRequiredPowerArtifacts, @@ -78,6 +81,13 @@ import { import { AsyncSemaphore } from './etl/async-semaphore'; import { discoverTraceReplayArtifacts } from './etl/trace-artifact-discovery'; import { discoverServerLogArtifacts, readServerLogArtifact } from './etl/server-log-artifacts'; +import { + discoverGpuMetricsArtifacts, + readPowerAuditValidations, +} from './etl/gpu-metrics-artifacts'; +import { createBenchmarkPowerAuditRecovery } from './etl/power-audit-recovery'; +import { ingestGpuMetricsArtifact } from './etl/gpu-metrics-ingest'; +import { readTelemetryReceipt, type TelemetryObservation } from './etl/telemetry-receipt'; import { datasetSlugFromBenchmarkRow } from './etl/dataset-provenance'; import { mapAggEvalRow, mapEvalRow } from './etl/eval-mapper'; import { ingestEvalRow } from './etl/eval-ingest'; @@ -96,6 +106,9 @@ import { const DEFAULT_REPO = 'SemiAnalysisAI/InferenceX'; const powerPublicationPoints = new Map(); const powerPublicationErrors: string[] = []; +const telemetryObservations = new Map(); +let telemetryExpectationsUnknown = false; +let checkTelemetry = false; const tracker = createSkipTracker(); const isDownloadMode = process.argv[2] === '--download'; @@ -151,6 +164,15 @@ if (isDownloadMode) { input.match(/^https:\/\/github\.com\/(?[^/]+\/[^/]+)\/actions\/runs\/\d+/u)?.[1] ?? DEFAULT_REPO; + runAttemptNum = fetchRunAttempt(REPO, runIdStr); + const requestedAttempt = input.match(/\/attempts\/(?\d+)/u)?.groups?.attempt; + if (requestedAttempt && Number(requestedAttempt) !== runAttemptNum) { + throw new Error( + `GitHub attempt ${runAttemptNum} differs from requested ${requestedAttempt}; ` + + 'use retained artifacts and exact source metadata for historical-attempt ingestion', + ); + } + // Download artifacts tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'ingest-')); artifactsDir = tempDir; @@ -194,8 +216,6 @@ if (isDownloadMode) { } console.log(`\n Downloaded ${byLogical.size} artifact(s)`); - - runAttemptNum = fetchRunAttempt(REPO, runIdStr); } else { // CI mode — read from env vars for (const key of [ @@ -286,6 +306,7 @@ async function main(): Promise { } validateRunBackfills(); + checkTelemetry = true; const configCache = createConfigCache(sql); const { getOrCreateConfig, preloadConfigs } = configCache; const { fetchGithubRun, getOrCreateWorkflowRun } = createWorkflowRunServices(sql, GITHUB_TOKEN, [ @@ -333,11 +354,15 @@ async function main(): Promise { } } - const requiredPowerPoints = verifyRequiredPowerArtifacts(artifactsDir, { - runId, - runAttempt: runAttemptNum, - headSha: ghInfo?.headSha ?? null, - }); + const requiredPowerPoints = verifyRequiredPowerArtifacts( + artifactsDir, + { + runId, + runAttempt: runAttemptNum, + headSha: ghInfo?.headSha ?? null, + }, + process.env.INGEST_REQUIRE_POWER === 'true', + ); if (requiredPowerPoints.length > 0) console.log(` Required power: ${requiredPowerPoints.length} source benchmark points verified`); @@ -409,6 +434,18 @@ async function main(): Promise { if (evalsOnly && requiredPowerPoints.length > 0) throw new Error('Required power: benchmark scope cannot be published as an evals-only run'); + if (requiredPowerPoints.length > 0) + await preflightRequiredPowerCurves( + sql, + artifactsDir, + { + runId, + runAttempt: runAttemptNum, + headSha: ghInfo?.headSha ?? null, + }, + { date, runStartedAt: workflowGhInfo?.runStartedAt ?? null, appendOnly }, + ); + const workflowRunId = await getOrCreateWorkflowRun({ githubRunId: runId, runAttempt: runAttemptNum, @@ -452,6 +489,8 @@ async function main(): Promise { let totalSampleFiles = 0; let totalChangelogs = 0; let totalTraceReplayLinked = 0; + let totalGpuMetricSeries = 0; + let totalGpuMetricSamples = 0; const datasetSlugs = new Set(); // Dataset slugs referenced by this run's agentic rows but absent from the // `datasets` table — timeline→dataset deep links 404 until they're ingested. @@ -484,6 +523,14 @@ async function main(): Promise { if (serverLogArtifacts.size > 0) { console.log(` Found ${serverLogArtifacts.size} server log artifact(s)`); } + // PowerX telemetry: `gpu_metrics_` is uploaded next to `bmk_` by + // every single-node job; multinode jobs carry it inside `power_audit_` + // instead (see migration 016). Digested here so the dashboard never + // re-downloads GitHub artifacts and keeps the series past retention. + const gpuMetricsArtifacts = discoverGpuMetricsArtifacts(artifactsDir); + if (gpuMetricsArtifacts.size > 0) { + console.log(` Found ${gpuMetricsArtifacts.size} telemetry artifact(s)`); + } // Sibling aiperf artifacts: each `bmk_agentic_` is paired with an // `agentic_` dir holding `profile_export.jsonl` and @@ -508,6 +555,7 @@ async function main(): Promise { const allBmkFiles = [...bmkFiles, ...allBmkDirs.flatMap((d) => findJsonFiles(d))]; const seenPointIdentities = new Map(); + const recoverPowerAudit = createBenchmarkPowerAuditRecovery(); console.log(` Found ${allBmkFiles.length} benchmark JSON file(s)`); for (const [fileIndex, file] of allBmkFiles.entries()) { @@ -519,6 +567,7 @@ async function main(): Promise { ); const data = readJson(file); if (!data) { + telemetryExpectationsUnknown = true; powerPublicationErrors.push(`Unreadable benchmark JSON: ${relativeFile}`); console.log(` skipped unreadable JSON (${elapsed(fileStart)})`); continue; @@ -535,16 +584,26 @@ async function main(): Promise { if (datasetSlug) datasetSlugs.add(datasetSlug); } + let fileExpectationsUnknown = false; const rows = rawRows - .filter((r) => typeof r === 'object' && r !== null) + .filter((r) => { + const isRow = typeof r === 'object' && r !== null; + if (!isRow) fileExpectationsUnknown = true; + return isRow; + }) .map((r) => { + const failedRuns = tracker.skips.failedRun; const mapped = mapBenchmarkRow(r, tracker, undefined, runIdStr); + // Known failed benchmarks are intentionally excluded; unmappable rows + // leave an unknown number of successful attachment expectations. + if (!mapped && tracker.skips.failedRun === failedRuns) fileExpectationsUnknown = true; if (!mapped && Number(r.isl) === 8192 && Number(r.osl) === 1024) { powerPublicationErrors.push(`Unmapped or failed 8K/1K result: ${relativeFile}`); } return mapped; }) .filter((r): r is NonNullable => r !== null); + if (fileExpectationsUnknown) telemetryExpectationsUnknown = true; console.log(` mapped rows: ${rows.length}`); if (rows.length === 0) { @@ -553,11 +612,34 @@ async function main(): Promise { } const toInsert = []; + const parentDir = path.basename(path.dirname(file)); + const configKey = parentDir.replace(/^bmk_/u, ''); + const suffix = stripBmkAndAgenticPrefix(parentDir); + const gpuMetricsArtifact = + gpuMetricsArtifacts.get(configKey) ?? gpuMetricsArtifacts.get(suffix); + let auditEvidence; + if (parentDir.startsWith('bmk_agentic_') && gpuMetricsArtifact) { + try { + auditEvidence = { + resultFile: path.basename(file), + validations: readPowerAuditValidations( + gpuMetricsArtifact.artifactDir, + gpuMetricsArtifact.artifactName, + ), + }; + } catch (error) { + tracker.recordTelemetryError( + `power audit for ${configKey}`, + error instanceof Error ? error : new Error(String(error)), + ); + } + } for (const row of rows) { let configId: number; try { configId = await getOrCreateConfig(row.config); } catch (error: any) { + telemetryExpectationsUnknown = true; tracker.recordDbError(`config for ${path.basename(file)}`, error); continue; } @@ -594,14 +676,32 @@ async function main(): Promise { `config ${configId}, conc ${row.conc}`, ); } + // Attach exact retained validation metadata before both the receipt and + // the upsert. A later aggregate copy must not null this run's recovery. + const point = recoverPowerAudit(applied.point, auditEvidence); const publication = powerPublicationPoint( - applied.point, + point, `https://github.com/${REPO}/actions/runs/${runIdNum}/attempts/${runAttemptNum}`, { path: relativeFile, sha256: artifactSha256 }, ); if (publication) powerPublicationPoints.set(publicationIdentity(publication.identity), publication); - toInsert.push(applied.point); + toInsert.push(point); + const identity = benchmarkPublicationIdentity(point); + const key = stablePowerPointIdentity(identity); + // Aggregate copies may arrive before/after their per-job sibling. Only + // the latter establishes whether that exact telemetry artifact exists. + if (parentDir.startsWith('bmk_') || !telemetryObservations.has(key)) { + telemetryObservations.set(key, { + identity, + artifactNames: gpuMetricsArtifact + ? [gpuMetricsArtifact.artifactName] + : parentDir.startsWith('bmk_') + ? [`gpu_metrics_${suffix}`, `power_audit_${suffix}`] + : [], + produced: parentDir.startsWith('bmk_') ? Boolean(gpuMetricsArtifact) : null, + }); + } } console.log(` rows with resolved configs: ${toInsert.length}`); @@ -637,14 +737,12 @@ async function main(): Promise { }); } - const parentDir = path.basename(path.dirname(file)); if (parentDir.startsWith('bmk_') && insertedIds.length > 0) { // Single-turn artifacts are `bmk_` paired with // `server_logs_`. Agentic artifacts are `bmk_agentic_` // but the server log is still `server_logs_` (no `agentic_` // prefix), so fall back to the fully-stripped suffix — otherwise // agentic rows never get their server log (and KV-pool size) linked. - const configKey = parentDir.replace(/^bmk_/u, ''); const logArtifact = serverLogArtifacts.get(configKey) ?? serverLogArtifacts.get(stripBmkAndAgenticPrefix(parentDir)); @@ -665,13 +763,48 @@ async function main(): Promise { tracker.recordDbError(`server_logs for ${configKey}`, error); } } + // Same pairing rule as server logs: `gpu_metrics_` carries no + // `agentic_` prefix, so agentic points fall back to the bare suffix. + if (gpuMetricsArtifact) { + try { + const gpuMetricsStart = Date.now(); + const ingested = await ingestGpuMetricsArtifact(sql, { + workflowRunId, + artifact: gpuMetricsArtifact, + benchmarkResultIds: insertedIds, + }); + if (ingested.seriesIds.length === 0) + throw new Error( + `No parseable telemetry series in ${gpuMetricsArtifact.artifactName}`, + ); + totalGpuMetricSeries += ingested.seriesIds.length; + totalGpuMetricSamples += ingested.samplesInserted; + console.log( + ` gpu_metrics ${ingested.seriesIds.length} series, ` + + `+${ingested.samplesInserted} sample(s), ` + + `${ingested.seriesSkipped} unchanged (${elapsed(gpuMetricsStart)})`, + ); + } catch (error: any) { + // Non-fatal on purpose: this point's benchmark rows are already + // committed and only its telemetry tab is affected, and + // `admin:db:backfill-gpu-metrics --run ` can re-digest the + // artifact later. Recording it as a DB error instead would reach + // the publication manifest and fail the whole production ingest. + tracker.recordTelemetryError(`gpu_metrics for ${configKey}`, error); + for (const row of toInsert) { + const point = telemetryObservations.get( + stablePowerPointIdentity(benchmarkPublicationIdentity(row)), + ); + if (point) point.error = error instanceof Error ? error.message : String(error); + } + } + } } // Trace-replay sibling lookup for agentic points only. The aiperf // harness emits `agentic_/trace_replay/...` next to the // `bmk_agentic_` artifact we just ingested. if (parentDir.startsWith('bmk_agentic_') && insertedIds.length > 0) { - const suffix = stripBmkAndAgenticPrefix(parentDir); const concMatch = path.basename(file).match(/_conc(?\d+)\.json$/u); const trace = (concMatch?.groups?.conc @@ -1042,9 +1175,24 @@ main() console.error('ingest-ci-run failed:', error); process.exitCode = 1; }) - .finally(() => { - const publicationPath = process.env.POWER_PUBLICATION_MANIFEST; + .finally(async () => { + const publicationPath = + process.env.POWER_PUBLICATION_MANIFEST ?? + `power-publication-${runIdNum}-attempt-${runAttemptNum}.json`; if (publicationPath) { + const telemetry = checkTelemetry + ? await readTelemetryReceipt( + sql, + { runId: runIdNum, runAttempt: runAttemptNum }, + [...telemetryObservations.values()], + { + expectedSource: + !telemetryExpectationsUnknown && telemetryObservations.size > 0 + ? 'benchmark_artifacts' + : 'unknown', + }, + ) + : undefined; fs.writeFileSync( publicationPath, `${JSON.stringify( @@ -1057,11 +1205,17 @@ main() ...powerPublicationErrors, ...(tracker.skips.dbError ? [`${tracker.skips.dbError} database ingest errors`] : []), ], + // Reported but not fatal — see Skips.telemetryError. + telemetryWarnings: tracker.skips.telemetryError + ? [`${tracker.skips.telemetryError} gpu_metrics digest errors`] + : [], + ...(telemetry ? { telemetry } : {}), }, null, 2, )}\n`, ); + console.log(` PowerX ingest receipt: ${publicationPath}`); } if (tempDir) { try { diff --git a/packages/db/src/ingest-gcs-backup.ts b/packages/db/src/ingest-gcs-backup.ts index 709fdc9c3..b9cb76b59 100644 --- a/packages/db/src/ingest-gcs-backup.ts +++ b/packages/db/src/ingest-gcs-backup.ts @@ -104,8 +104,12 @@ interface WorkflowMapResult { changelogs: { baseRef: string; headRef: string; entries: ChangelogEntry[] }[]; /** True when the changelog declares evals-only — benchmark/stats data is dropped. */ evalsOnly: boolean; - /** Skip counts from mapping phase (dbError is tracked separately in phase 2). */ - localSkips: Omit; + /** + * Skip counts from the mapping phase. `dbError` is tracked separately in phase 2, + * and `telemetryError` never applies here: the GCS backup path ingests no + * `gpu_metrics_*` artifacts. + */ + localSkips: Omit; localUnmappedModels: Set; localUnmappedHws: Set; /** Pre-formatted [WARN] lines to print at the start of phase 2 for this dir. */ @@ -121,7 +125,7 @@ interface WriteResult { evalSamples: number; changelogs: number; warnings: string[]; - localSkips: Omit; + localSkips: Omit; localUnmappedModels: string[]; localUnmappedHws: string[]; } diff --git a/packages/db/src/lib/artifact-retry.ts b/packages/db/src/lib/artifact-retry.ts index e145aa04d..4044dc35b 100644 --- a/packages/db/src/lib/artifact-retry.ts +++ b/packages/db/src/lib/artifact-retry.ts @@ -1,5 +1,17 @@ const DEFAULT_RETRY_DELAYS_MS = [5_000, 15_000, 30_000, 60_000, 120_000] as const; +/** + * Marks a failure no retry can fix (the run or artifact is gone, not + * unreachable). `retryArtifactOperation` rethrows it immediately instead of + * burning the full backoff schedule on it. + */ +export class NonRetryableArtifactError extends Error { + constructor(message: string) { + super(message); + this.name = 'NonRetryableArtifactError'; + } +} + interface ArtifactRetryOptions { delaysMs?: readonly number[]; wait?: (delayMs: number) => Promise; @@ -26,6 +38,7 @@ export async function retryArtifactOperation( try { return await operation(); } catch (error) { + if (error instanceof NonRetryableArtifactError) throw error; lastError = error; if (attempt >= delaysMs.length) break; const delayMs = delaysMs[attempt]; diff --git a/packages/db/src/lib/backfill-benchmark-refresh.ts b/packages/db/src/lib/backfill-benchmark-refresh.ts new file mode 100644 index 000000000..15d2d60f6 --- /dev/null +++ b/packages/db/src/lib/backfill-benchmark-refresh.ts @@ -0,0 +1,224 @@ +import { isDeepStrictEqual } from 'node:util'; +import { DB_MODEL_TO_DISPLAY } from '@semianalysisai/inferencex-constants'; +import { refreshLatestBenchmarks, type Sql } from '../etl/db-utils.js'; +import { + stablePowerPointIdentity, + type PowerPublicationManifest, +} from '../etl/power-publication.js'; + +export type BenchmarkAuditUpdate = NonNullable< + NonNullable['auditUpdates'] +>[number]; + +function refreshTarget(manifest: PowerPublicationManifest) { + const raw = process.env.CACHE_INVALIDATE_URL; + if (!raw) throw new Error('CACHE_INVALIDATE_URL is required for benchmark metadata refresh'); + const endpoint = new URL(raw); + if ( + !['http:', 'https:'].includes(endpoint.protocol) || + endpoint.pathname !== '/api/v1/invalidate' || + endpoint.search || + endpoint.hash || + endpoint.username || + endpoint.password + ) + throw new Error( + 'CACHE_INVALIDATE_URL must name /api/v1/invalidate without credentials or query', + ); + const previous = manifest.benchmarkRefresh?.endpoint; + if (previous && previous !== endpoint.href) + throw new Error('Benchmark refresh receipt target differs from CACHE_INVALIDATE_URL'); + const secret = process.env.CACHE_INVALIDATE_SECRET || process.env.INVALIDATE_SECRET; + if (!secret) throw new Error('CACHE_INVALIDATE_SECRET or INVALIDATE_SECRET is required'); + return { + endpoint, + headers: { + Authorization: `Bearer ${secret}`, + ...(process.env.CACHE_PROTECTION_BYPASS_SECRET + ? { 'x-vercel-protection-bypass': process.env.CACHE_PROTECTION_BYPASS_SECRET } + : {}), + }, + }; +} + +/** Persist possible writes before ingest, including a crash before UPDATE returns its IDs. */ +export function checkpointBenchmarkRefresh( + manifest: PowerPublicationManifest, + ids: number[], + save: () => void, + auditUpdates: BenchmarkAuditUpdate[] = [], +): void { + if (ids.length === 0) return; + const previous = manifest.benchmarkRefresh; + manifest.benchmarkRefresh = { + status: 'pending', + benchmarkResultIds: [ + ...new Set([ + ...(previous?.status === 'complete' ? [] : (previous?.benchmarkResultIds ?? [])), + ...ids, + ]), + ], + auditUpdates: [...(previous?.auditUpdates ?? [])], + ...(previous?.endpoint ? { endpoint: previous.endpoint } : {}), + }; + for (const update of auditUpdates) { + const existing = manifest.benchmarkRefresh.auditUpdates!.find( + (value) => value.benchmarkResultId === update.benchmarkResultId, + ); + if ( + existing && + (!isDeepStrictEqual(existing.powerAudit, update.powerAudit) || + stablePowerPointIdentity(existing.identity) !== stablePowerPointIdentity(update.identity)) + ) + throw new Error('Conflicting retained audit evidence for benchmark refresh'); + if (!existing) manifest.benchmarkRefresh.auditUpdates!.push(update); + } + save(); + try { + manifest.benchmarkRefresh.endpoint = refreshTarget(manifest).endpoint.href; + } catch (error) { + manifest.benchmarkRefresh.status = 'failed'; + manifest.benchmarkRefresh.error = error instanceof Error ? error.message : String(error); + throw error; + } finally { + save(); + } +} + +/** Retry only the materialized view/cache/API phase; never download or rewrite samples. */ +export async function refreshBackfillBenchmarks( + sql: Sql, + manifest: PowerPublicationManifest, + save: () => void, +): Promise { + const receipt = manifest.benchmarkRefresh; + if (!receipt) throw new Error('Receipt has no benchmark metadata refresh responsibility'); + receipt.status = 'pending'; + receipt.checkedAt = new Date().toISOString(); + delete receipt.error; + save(); + let phase = 'validate target'; + try { + const { endpoint, headers } = refreshTarget(manifest); + receipt.endpoint = endpoint.href; + save(); + const ids = receipt.benchmarkResultIds; + if ( + !Array.isArray(ids) || + ids.some((id) => !Number.isSafeInteger(id) || id <= 0) || + new Set(ids).size !== ids.length + ) + throw new Error('Invalid benchmarkResultIds in refresh receipt'); + const rows = await sql< + (Record & { id: number; model: string; power_audit: unknown })[] + >` + select c.*, br.id, br.benchmark_type, br.isl, br.osl, br.conc, br.offload_mode, + br.recipe_fingerprint, br.power_audit from benchmark_results br + join configs c on c.id = br.config_id + join workflow_runs wr on wr.id = br.workflow_run_id + where wr.github_run_id = ${manifest.runId} and wr.run_attempt = ${manifest.runAttempt} + and br.id = any(${sql.array(ids)}::bigint[]) + `; + if (rows.length !== ids.length) + throw new Error('Refresh receipt IDs do not belong to its run/attempt'); + const changed = rows; + for (const row of changed) { + const audit = row.power_audit; + if (audit === null) + throw new Error( + `Benchmark ${row.id} power_audit is NULL; targeted artifact re-ingest is required before cache refresh`, + ); + if ( + !audit || + typeof audit !== 'object' || + Array.isArray(audit) || + !('source' in audit) || + typeof audit.source !== 'string' || + !audit.source.trim() + ) + throw new Error(`Benchmark ${row.id} power_audit must be an object with a nonempty source`); + const updates = + receipt.auditUpdates?.filter((update) => update.benchmarkResultId === Number(row.id)) ?? []; + if (updates.length !== 1 || !isDeepStrictEqual(updates[0]!.powerAudit, audit)) + throw new Error(`Benchmark ${row.id} power_audit differs from retained source evidence`); + const update = updates[0]!; + if (stablePowerPointIdentity(update.identity) !== stablePowerPointIdentity(row)) + throw new Error(`Benchmark ${row.id} identity differs from retained source mapping`); + const identities = new Set( + [update.identity, update.sourceIdentity ?? update.identity].map(stablePowerPointIdentity), + ); + const points = manifest.points.filter((point) => + identities.has(stablePowerPointIdentity(point.identity)), + ); + if (points.length > 1) + throw new Error(`Ambiguous publication identity for benchmark ${row.id}`); + if ( + points.length === 1 && + (points[0]!.power_audit === null || points[0]!.power_audit === undefined) + ) + points[0]!.power_audit = update.powerAudit; + } + save(); + if (changed.length > 0) { + phase = 'refresh latest_benchmarks'; + await refreshLatestBenchmarks(sql); + phase = 'invalidate cache'; + const response = await fetch(endpoint, { + method: 'POST', + headers, + redirect: 'error', + signal: AbortSignal.timeout(30_000), + }); + if (!response.ok) throw new Error(`HTTP ${response.status}`); + const result = await response.json(); + if ( + result?.invalidated !== true || + !Number.isSafeInteger(result.blobsDeleted) || + result.blobsDeleted < 0 + ) + throw new Error('Invalid cache invalidation response'); + phase = 'verify benchmark API'; + for (const model of new Set(changed.map((row) => row.model))) { + const displayModel = DB_MODEL_TO_DISPLAY[model]; + if (!displayModel) throw new Error(`Unmapped public model: ${model}`); + const url = new URL('/api/v1/benchmarks', endpoint); + url.search = new URLSearchParams({ + model: displayModel, + runId: String(manifest.runId), + exactRun: 'true', + }).toString(); + const api = await fetch(url, { + headers: process.env.CACHE_PROTECTION_BYPASS_SECRET + ? { 'x-vercel-protection-bypass': process.env.CACHE_PROTECTION_BYPASS_SECRET } + : undefined, + redirect: 'error', + signal: AbortSignal.timeout(30_000), + }); + if (!api.ok) throw new Error(`HTTP ${api.status}`); + const body: unknown = await api.json(); + if (!Array.isArray(body)) throw new Error('Invalid benchmark API response'); + for (const row of changed.filter((entry) => entry.model === model)) { + const matches = body.filter((candidate) => { + const raw = candidate?.id; + const id = + typeof raw === 'number' + ? raw + : typeof raw === 'string' && /^[1-9]\d*$/u.test(raw) + ? Number(raw) + : NaN; + return Number.isSafeInteger(id) && id > 0 && id === Number(row.id); + }); + if (matches.length !== 1 || !isDeepStrictEqual(matches[0].power_audit, row.power_audit)) + throw new Error(`Benchmark ${row.id} power_audit differs from database`); + } + } + } + receipt.status = 'complete'; + } catch (error) { + receipt.status = 'failed'; + receipt.error = `${phase}: ${error instanceof Error ? error.message : String(error)}`; + throw new Error(receipt.error, { cause: error }); + } finally { + save(); + } +} diff --git a/packages/db/src/lib/backfill-runner.ts b/packages/db/src/lib/backfill-runner.ts index 876e559b2..2051e493d 100644 --- a/packages/db/src/lib/backfill-runner.ts +++ b/packages/db/src/lib/backfill-runner.ts @@ -8,6 +8,12 @@ import { confirm, hasYesFlag } from '../cli-utils.js'; import type { Sql } from '../etl/db-utils.js'; +import { retryArtifactOperation } from './artifact-retry.js'; +import { + WorkflowRunNotFoundError, + listRunArtifacts, + type ArtifactMeta, +} from './github-artifacts.js'; export interface LimitForceFlags { limit: number | null; @@ -165,6 +171,25 @@ export async function runCandidateIdBackfill( * class instances/prototypes so postgres.js serializes plain data only — * matches what the inline ingest path stores. */ +/** + * List a candidate run's GitHub artifacts with transient-failure retry. + * Returns `null` when GitHub no longer has the run at all, so a sweep over + * months of history reports the gap and moves on instead of aborting. + */ +export async function listBackfillRunArtifacts( + repository: string, + runId: number, +): Promise { + try { + return await retryArtifactOperation(`listing GitHub artifacts for run ${runId}`, () => + listRunArtifacts(repository, String(runId)), + ); + } catch (error) { + if (error instanceof WorkflowRunNotFoundError) return null; + throw error; + } +} + export function jsonbParam(sql: Sql, value: unknown): ReturnType { return sql.json(structuredClone(value) as unknown as Parameters[0]); } diff --git a/packages/db/src/lib/benchmark-result-lookup.test.ts b/packages/db/src/lib/benchmark-result-lookup.test.ts new file mode 100644 index 000000000..15478e635 --- /dev/null +++ b/packages/db/src/lib/benchmark-result-lookup.test.ts @@ -0,0 +1,164 @@ +import fs from 'node:fs'; + +import { PGlite } from '@electric-sql/pglite'; +import type postgres from 'postgres'; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from 'vitest'; + +import { mapBenchmarkRow, type BenchmarkParams } from '../etl/benchmark-mapper'; +import { createSkipTracker } from '../etl/skip-tracker'; +import { filterPurgedBenchmarkRows, findBenchmarkResultIds } from './benchmark-result-lookup'; +import { collectMissingTelemetryExpectations } from './gpu-metrics-backfill'; + +type Sql = postgres.Sql; +let db: PGlite; +let sql: Sql; +const roots: string[] = []; + +function queryClient(database: Pick) { + const client = async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await database.query(query, values); + return result.rows; + }; + return Object.assign(client, { json: JSON.stringify, array: (value: unknown) => value }); +} + +const GITHUB_RUN_ID = 34557177019; +const RUN = { github_run_id: GITHUB_RUN_ID, run_attempt: 2 }; + +/** Raw `bmk_` JSON as the runner emits it for a fixed-sequence sweep. */ +function rawRow(overrides: Record = {}): Record { + return { + infmax_model_prefix: 'dsr1', + hw: 'b200-nv', + framework: 'sglang', + precision: 'fp4', + isl: 8192, + osl: 1024, + conc: 32, + tp: 8, + ep: 1, + dp_attention: false, + tput_per_gpu: 1234.5, + ...overrides, + }; +} + +function mapped(raw: Record): BenchmarkParams { + const row = mapBenchmarkRow(raw, createSkipTracker()); + if (!row) throw new Error('fixture did not map'); + return row; +} + +beforeAll(async () => { + db = await PGlite.create(); + const dir = new URL('../../migrations/', import.meta.url); + for (const name of fs.readdirSync(dir).toSorted()) { + if (name.endsWith('.sql')) await db.exec(fs.readFileSync(new URL(name, dir), 'utf8')); + } + sql = queryClient(db) as unknown as Sql; +}, 30_000); + +afterEach(() => { + for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); +}); + +afterAll(async () => { + await db?.close(); +}); + +/** + * Config 1 matches the fixture rows; config 2 differs only in decode_tp. Attempt 1 + * of the same GitHub run holds a point that must never be matched from attempt 2. + */ +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, created_at, date) + VALUES (1, ${GITHUB_RUN_ID}, 1, 'Run Sweep', '2026-09-11T04:00:00Z', '2026-09-11'), + (2, ${GITHUB_RUN_ID}, 2, 'Run Sweep', '2026-09-11T06:00:00Z', '2026-09-11'); + + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + is_multinode, prefill_tp, prefill_ep, prefill_dp_attention, prefill_num_workers, + decode_tp, decode_ep, decode_dp_attention, decode_num_workers, + num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, false, 8, 1, false, 0, + 8, 1, false, 0, 8, 8), + (2, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, false, 8, 1, false, 0, + 4, 1, false, 0, 8, 8), + (3, 'dsv4', 'b200', 'vllm', 'fp4', 'none', false, false, 1, 1, false, 0, + 1, 1, false, 0, 1, 1); + + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, + isl, osl, conc, offload_mode, metrics) + VALUES (20, 2, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, 'off', '{}'), + (21, 2, 1, 'single_turn', '2026-09-11', 8192, 1024, 64, 'off', '{}'), + (22, 2, 2, 'single_turn', '2026-09-11', 8192, 1024, 32, 'off', '{}'), + (23, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, 'off', '{}'), + (24, 2, 3, 'agentic_traces', '2026-09-11', null, null, 72, 'off', '{}');`); +}); + +describe('findBenchmarkResultIds', () => { + it('prefers the offload-exact point and reports no unique fallback', async () => { + await db.exec(`INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, + date, isl, osl, conc, offload_mode, metrics) + VALUES (25, 2, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, 'on', '{}');`); + const fallbacks: number[] = []; + const ids = await findBenchmarkResultIds(sql, RUN, [mapped(rawRow())], (id) => + fallbacks.push(id), + ); + expect(ids).toEqual([20]); + expect(fallbacks).toEqual([]); + }); +}); + +it('excludes exact purges from mixed artifacts and missing-artifact expectations', async () => { + await db.exec(`UPDATE benchmark_results SET config_id = 1 WHERE id = 24; + UPDATE configs SET id = 744 WHERE id = 3; + UPDATE workflow_runs SET github_run_id = 30730541420, run_attempt = 1 WHERE id = 2; + UPDATE benchmark_results SET config_id = 744, conc = 1, offload_mode = 'on' WHERE id = 24;`); + const run = { github_run_id: 30730541420, run_attempt: 1 }; + const point: BenchmarkParams = { + ...mapped(rawRow({ infmax_model_prefix: 'dsv4', framework: 'vllm', tp: 1 })), + benchmarkType: 'agentic_traces', + isl: null, + osl: null, + offloadMode: 'on', + recipeFingerprint: null, + }; + const purged = { ...point, conc: 32 }; + const surviving = { ...point, conc: 1 }; + const rows = await filterPurgedBenchmarkRows(sql, run, [purged, surviving]); + expect(rows).toEqual([surviving]); + expect(await findBenchmarkResultIds(sql, run, rows)).toEqual([24]); + expect(await filterPurgedBenchmarkRows(sql, run, [purged])).toEqual([]); + expect(await filterPurgedBenchmarkRows(sql, { ...run, run_attempt: 2 }, [purged])).toEqual([ + purged, + ]); + expect( + await filterPurgedBenchmarkRows(sql, run, [{ ...purged, recipeFingerprint: 'new' }]), + ).toHaveLength(1); + expect( + await filterPurgedBenchmarkRows(sql, run, [{ ...purged, offloadMode: 'off' }]), + ).toHaveLength(1); + expect( + await filterPurgedBenchmarkRows(sql, run, [ + { ...purged, config: { ...point.config, decodeTp: 4 } }, + ]), + ).toHaveLength(1); + const missing = await collectMissingTelemetryExpectations( + [ + { + id: 1, + name: 'bmk_agentic_retained', + expired: false, + created_at: '2026-09-11T00:00:00Z', + archive_download_url: 'https://example.test/1', + }, + ], + [], + null, + () => filterPurgedBenchmarkRows(sql, run, [purged, surviving]), + ); + expect(missing.observations.map((entry) => entry.identity.conc)).toEqual([1]); + expect(missing.errors).toEqual([]); +}); diff --git a/packages/db/src/lib/benchmark-result-lookup.ts b/packages/db/src/lib/benchmark-result-lookup.ts new file mode 100644 index 000000000..f94128f4f --- /dev/null +++ b/packages/db/src/lib/benchmark-result-lookup.ts @@ -0,0 +1,153 @@ +/** + * Resolve persisted `benchmark_results` ids for raw artifact rows that were + * mapped through the production mapper. Shared by the sidecar backfills + * (server logs, gpu_metrics) so every historical attachment uses the same + * natural-key match as the CI ingest path. + */ + +import fs from 'node:fs'; +import path from 'node:path'; + +import { mapBenchmarkRow, type BenchmarkParams } from '../etl/benchmark-mapper.js'; +import { createSkipTracker } from '../etl/skip-tracker.js'; +import type { Sql } from '../etl/db-utils.js'; +import { configCacheKey } from '../etl/config-cache.js'; +import { isBenchmarkPointPurged, PURGED_BENCHMARK_POINTS } from '../etl/run-overrides.js'; +import { resolveServerLogResultCandidates } from './server-log-backfill.js'; + +export interface BenchmarkRunSelector { + github_run_id: number; + run_attempt: number; +} + +/** Purged points must not become missing-result errors or receipt expectations. */ +export async function filterPurgedBenchmarkRows( + sql: Sql, + run: BenchmarkRunSelector, + rows: readonly BenchmarkParams[], +): Promise { + const configIds = PURGED_BENCHMARK_POINTS.filter( + (point) => point.githubRunId === run.github_run_id && point.runAttempt === run.run_attempt, + ).map((point) => point.configId); + if (configIds.length === 0) return [...rows]; + + const configs = new Map(); + const retained: BenchmarkParams[] = []; + for (const row of rows) { + const c = row.config; + const key = configCacheKey(c); + if (!configs.has(key)) { + const [config] = await sql<{ id: number }[]>` + select id from configs cfg + where cfg.id = any(${configIds}::integer[]) + and cfg.hardware = ${c.hardware} and cfg.framework = ${c.framework} + and cfg.model = ${c.model} and cfg.precision = ${c.precision} + and cfg.spec_method = ${c.specMethod} and cfg.disagg = ${c.disagg} + and cfg.is_multinode = ${c.isMultinode} + and cfg.prefill_tp = ${c.prefillTp} and cfg.prefill_ep = ${c.prefillEp} + and cfg.prefill_dp_attention = ${c.prefillDpAttn} + and cfg.prefill_num_workers = ${c.prefillNumWorkers} + and cfg.decode_tp = ${c.decodeTp} and cfg.decode_ep = ${c.decodeEp} + and cfg.decode_dp_attention = ${c.decodeDpAttn} + and cfg.decode_num_workers = ${c.decodeNumWorkers} + and cfg.num_prefill_gpu = ${c.numPrefillGpu} + and cfg.num_decode_gpu = ${c.numDecodeGpu} + `; + configs.set(key, config ? Number(config.id) : null); + } + const configId = configs.get(key)!; + if ( + configId === null || + !isBenchmarkPointPurged(run.github_run_id, run.run_attempt, { ...row, configId }) + ) + retained.push(row); + } + return retained; +} + +export async function findBenchmarkResultIds( + sql: Sql, + run: BenchmarkRunSelector, + rows: readonly BenchmarkParams[], + onUniqueFallback: (id: number) => void = () => {}, +): Promise { + const ids = new Set(); + for (const row of rows) { + const c = row.config; + const candidates = await sql<{ id: number; offload_mode: string }[]>` + select br.id, br.offload_mode + from benchmark_results br + join workflow_runs wr on wr.id = br.workflow_run_id + join configs cfg on cfg.id = br.config_id + where wr.github_run_id = ${run.github_run_id} + and wr.run_attempt = ${run.run_attempt} + and cfg.hardware = ${c.hardware} + and cfg.framework = ${c.framework} + and cfg.model = ${c.model} + and cfg.precision = ${c.precision} + and cfg.spec_method = ${c.specMethod} + and cfg.disagg = ${c.disagg} + and cfg.is_multinode = ${c.isMultinode} + and cfg.prefill_tp = ${c.prefillTp} + and cfg.prefill_ep = ${c.prefillEp} + and cfg.prefill_dp_attention = ${c.prefillDpAttn} + and cfg.prefill_num_workers = ${c.prefillNumWorkers} + and cfg.decode_tp = ${c.decodeTp} + and cfg.decode_ep = ${c.decodeEp} + and cfg.decode_dp_attention = ${c.decodeDpAttn} + and cfg.decode_num_workers = ${c.decodeNumWorkers} + and cfg.num_prefill_gpu = ${c.numPrefillGpu} + and cfg.num_decode_gpu = ${c.numDecodeGpu} + and br.benchmark_type = ${row.benchmarkType} + and br.isl is not distinct from ${row.isl} + and br.osl is not distinct from ${row.osl} + and br.conc = ${row.conc} + and br.recipe_fingerprint is not distinct from ${row.recipeFingerprint} + `; + const resolution = resolveServerLogResultCandidates( + candidates.map((candidate) => ({ + id: Number(candidate.id), + offloadMode: candidate.offload_mode, + })), + row.offloadMode, + ); + if (resolution.usedUniqueFallback) onUniqueFallback(resolution.ids[0]!); + for (const id of resolution.ids) ids.add(id); + } + return [...ids]; +} + +function findJsonFiles(root: string): string[] { + const files: string[] = []; + for (const entry of fs.readdirSync(root, { withFileTypes: true })) { + const pathname = path.join(root, entry.name); + if (entry.isDirectory()) files.push(...findJsonFiles(pathname)); + else if (entry.isFile() && entry.name.endsWith('.json')) files.push(pathname); + } + return files.toSorted(); +} + +/** Map known successful rows; report unidentifiable rows separately from known failures. */ +export function readMappedBenchmarkRows( + root: string, + onUnmapped: (error: string) => void = () => {}, +): BenchmarkParams[] { + const tracker = createSkipTracker(); + const rows: BenchmarkParams[] = []; + for (const file of findJsonFiles(root)) { + const parsed = JSON.parse(fs.readFileSync(file, 'utf8')) as unknown; + const rawRows = Array.isArray(parsed) ? parsed : [parsed]; + for (const [index, raw] of rawRows.entries()) { + const error = `Unmappable benchmark row: ${path.relative(root, file)} row ${index + 1}`; + if (!raw || typeof raw !== 'object' || Array.isArray(raw)) { + onUnmapped(error); + continue; + } + const failedRuns = tracker.skips.failedRun; + const mapped = mapBenchmarkRow(raw as Record, tracker); + if (mapped) rows.push(mapped); + else if (tracker.skips.failedRun === failedRuns) onUnmapped(error); + } + } + return rows; +} diff --git a/packages/db/src/lib/github-artifacts.test.ts b/packages/db/src/lib/github-artifacts.test.ts index 571643a5e..e612d5c1b 100644 --- a/packages/db/src/lib/github-artifacts.test.ts +++ b/packages/db/src/lib/github-artifacts.test.ts @@ -1,5 +1,4 @@ import { describe, expect, it } from 'vitest'; - import { RUNNER_SUFFIX_RE, dedupeArtifactsByLogicalName } from './github-artifacts.js'; const art = (name: string, created_at: string) => ({ diff --git a/packages/db/src/lib/github-artifacts.ts b/packages/db/src/lib/github-artifacts.ts index 67827855e..c95957755 100644 --- a/packages/db/src/lib/github-artifacts.ts +++ b/packages/db/src/lib/github-artifacts.ts @@ -8,6 +8,8 @@ import { execSync } from 'node:child_process'; import fs from 'node:fs'; import path from 'node:path'; +import { NonRetryableArtifactError } from './artifact-retry.js'; + export interface ArtifactMeta { id?: number; name: string; @@ -32,12 +34,46 @@ export interface ArtifactMeta { */ export const RUNNER_SUFFIX_RE = /_[a-zA-Z][a-zA-Z0-9.-]*_\d+$/u; -/** List a workflow run's artifacts via `gh api` (paginated). Malformed lines are skipped. */ +/** + * GitHub no longer has the run (deleted, or purged past retention), so its + * artifact listing 404s. Distinct from expired artifacts, which still list + * with `expired: true`. + */ +export class WorkflowRunNotFoundError extends NonRetryableArtifactError { + readonly repo: string; + readonly runId: string; + + constructor(repo: string, runId: string) { + super(`GitHub has no run ${runId} in ${repo} (HTTP 404)`); + this.name = 'WorkflowRunNotFoundError'; + this.repo = repo; + this.runId = runId; + } +} + +/** `gh api` reports the HTTP status in its stderr (`gh: Not Found (HTTP 404)`). */ +export function isGithubNotFoundError(error: unknown): boolean { + if (typeof error !== 'object' || error === null) return false; + const { stderr, message } = error as { stderr?: unknown; message?: unknown }; + const text = [stderr, message].filter((part) => typeof part === 'string').join('\n'); + return /\(HTTP 404\)/u.test(text); +} + +/** + * List a workflow run's artifacts via `gh api` (paginated). Malformed lines + * are skipped; a 404 surfaces as `WorkflowRunNotFoundError`. + */ export function listRunArtifacts(repo: string, runId: string): ArtifactMeta[] { - const json = execSync( - `gh api "repos/${repo}/actions/runs/${runId}/artifacts" --paginate --jq '.artifacts[]'`, - { encoding: 'utf8', maxBuffer: 50 * 1024 * 1024 }, - ); + let json: string; + try { + json = execSync( + `gh api "repos/${repo}/actions/runs/${runId}/artifacts" --paginate --jq '.artifacts[]'`, + { encoding: 'utf8', maxBuffer: 50 * 1024 * 1024 }, + ); + } catch (error) { + if (isGithubNotFoundError(error)) throw new WorkflowRunNotFoundError(repo, runId); + throw error; + } const out: ArtifactMeta[] = []; for (const line of json.trim().split('\n')) { if (!line) continue; diff --git a/packages/db/src/lib/gpu-metric-stats.ts b/packages/db/src/lib/gpu-metric-stats.ts new file mode 100644 index 000000000..63bd4ec2a --- /dev/null +++ b/packages/db/src/lib/gpu-metric-stats.ts @@ -0,0 +1,61 @@ +import { + computeGpuMetricStats, + type GpuMetricSample, + type GpuMetricStats, +} from '../etl/gpu-metrics-csv'; + +/** Bump whenever full-record digest definitions change. Independent of serving-window metrics. */ +export const GPU_STATS_VERSION = 1; + +export const STAT_METRIC_COLUMN = { + powerW: 'power_w', + temperatureC: 'temperature_c', + smClockMhz: 'sm_clock_mhz', + memClockMhz: 'mem_clock_mhz', + gpuUtilPct: 'gpu_util_pct', + memUtilPct: 'mem_util_pct', + edgeTempC: 'edge_temp_c', + memTempC: 'mem_temp_c', + gfxVoltageMv: 'gfx_voltage_mv', + socVoltageMv: 'soc_voltage_mv', + memVoltageMv: 'mem_voltage_mv', + fclkMhz: 'fclk_mhz', + socclkMhz: 'socclk_mhz', + mmActivityPct: 'mm_activity_pct', +} as const satisfies Record; + +/** Stored metric names use the column spelling so SQL readers need no mapping. */ +export function statMetricColumn(metric: GpuMetricStats['metric']): string { + return STAT_METRIC_COLUMN[metric]; +} + +export type StoredGpuMetricSample = { + gpu_index: number; + sampled_at: string | Date; +} & Record<(typeof STAT_METRIC_COLUMN)[keyof typeof STAT_METRIC_COLUMN], number | null>; + +/** Preserve nulls: chart adapters intentionally zero-fill power, digest inputs must not. */ +export function computeStoredGpuMetricStats( + rows: readonly StoredGpuMetricSample[], +): GpuMetricStats[] { + return computeGpuMetricStats( + rows.map((row): GpuMetricSample => ({ + timestampMs: new Date(row.sampled_at).getTime(), + gpuIndex: Number(row.gpu_index), + powerW: row.power_w, + temperatureC: row.temperature_c, + smClockMhz: row.sm_clock_mhz, + memClockMhz: row.mem_clock_mhz, + gpuUtilPct: row.gpu_util_pct, + memUtilPct: row.mem_util_pct, + edgeTempC: row.edge_temp_c, + memTempC: row.mem_temp_c, + gfxVoltageMv: row.gfx_voltage_mv, + socVoltageMv: row.soc_voltage_mv, + memVoltageMv: row.mem_voltage_mv, + fclkMhz: row.fclk_mhz, + socclkMhz: row.socclk_mhz, + mmActivityPct: row.mm_activity_pct, + })), + ); +} diff --git a/packages/db/src/lib/gpu-metrics-backfill.test.ts b/packages/db/src/lib/gpu-metrics-backfill.test.ts new file mode 100644 index 000000000..bb9607ca7 --- /dev/null +++ b/packages/db/src/lib/gpu-metrics-backfill.test.ts @@ -0,0 +1,50 @@ +import { describe, expect, it } from 'vitest'; + +import type { ArtifactMeta } from './github-artifacts.js'; +import { pairGpuMetricsArtifacts } from './gpu-metrics-backfill.js'; + +const meta = ( + name: string, + id: number, + created_at = '2026-09-11T00:00:00Z', + expired = false, +): ArtifactMeta => ({ + id, + name, + created_at, + expired, + archive_download_url: `https://example.test/${id}`, +}); + +describe('pairGpuMetricsArtifacts', () => { + it('lets a power_audit bundle stand in only when its suffix has no gpu_metrics upload', () => { + // The bundle is listed before its gpu_metrics sibling on purpose: input + // order alone must not decide the winner. + const pairs = pairGpuMetricsArtifacts([ + meta('power_audit_cfg-mn_b200-slurm_0', 1), + meta('bmk_cfg-mn_b200-slurm_0', 2), + meta('power_audit_cfg-sn_h200-cw_0', 4), + meta('gpu_metrics_cfg-sn_h200-cw_0', 3), + meta('bmk_cfg-sn_h200-cw_0', 5), + meta('power_audit_orphan_b200-slurm_0', 6), + ]); + expect(pairs.map((pair) => [pair.gpuMetrics.name, pair.benchmarks.name])).toEqual([ + ['gpu_metrics_cfg-sn_h200-cw_0', 'bmk_cfg-sn_h200-cw_0'], + ['power_audit_cfg-mn_b200-slurm_0', 'bmk_cfg-mn_b200-slurm_0'], + ]); + }); + + it('keeps only the newest retry per logical benchmark and drops expired uploads', () => { + const pairs = pairGpuMetricsArtifacts([ + meta('gpu_metrics_cfg-a_h200-cw_0', 1, '2026-09-11T00:00:00Z'), + meta('bmk_cfg-a_h200-cw_0', 2, '2026-09-11T00:00:00Z'), + meta('gpu_metrics_cfg-a_h200-dgxc-slurm_1', 3, '2026-09-11T02:00:00Z'), + meta('bmk_cfg-a_h200-dgxc-slurm_1', 4, '2026-09-11T02:00:00Z'), + meta('gpu_metrics_cfg-c_h200-cw_0', 5, '2026-09-11T03:00:00Z', true), + meta('bmk_cfg-c_h200-cw_0', 6, '2026-09-11T03:00:00Z'), + ]); + expect(pairs.map((pair) => pair.gpuMetrics.name)).toEqual([ + 'gpu_metrics_cfg-a_h200-dgxc-slurm_1', + ]); + }); +}); diff --git a/packages/db/src/lib/gpu-metrics-backfill.ts b/packages/db/src/lib/gpu-metrics-backfill.ts new file mode 100644 index 000000000..d1600e2d3 --- /dev/null +++ b/packages/db/src/lib/gpu-metrics-backfill.ts @@ -0,0 +1,134 @@ +import type { Sql } from '../etl/db-utils.js'; +import { GPU_STATS_VERSION } from './gpu-metric-stats.js'; + +/** Pairing rules for the historical gpu_metrics backfill. */ + +import { gpuMetricsArtifactSuffix, isPowerAuditArtifact } from '../etl/gpu-metrics-artifacts.js'; +import type { BenchmarkParams } from '../etl/benchmark-mapper.js'; +import { benchmarkPublicationIdentity } from '../etl/power-publication.js'; +import type { TelemetryObservation, TelemetryReceipt } from '../etl/telemetry-receipt.js'; +import { + dedupeArtifactsByLogicalName, + RUNNER_SUFFIX_RE, + type ArtifactMeta, +} from './github-artifacts.js'; + +export interface GpuMetricsArtifactPair { + /** `gpu_metrics_`, or the `power_audit_` bundle when a multinode job uploaded no other. */ + gpuMetrics: ArtifactMeta; + benchmarks: ArtifactMeta; +} + +function isNewerArtifact(candidate: ArtifactMeta, existing: ArtifactMeta): boolean { + return ( + candidate.created_at > existing.created_at || + (candidate.created_at === existing.created_at && (candidate.id ?? 0) > (existing.id ?? 0)) + ); +} + +/** + * Pair every unexpired telemetry artifact with its exact bmk sibling. Retried + * jobs upload on different runners, so the newest artifact per logical + * (runner-suffix-stripped) benchmark name wins, matching CI ingest. A + * `power_audit_` bundle pairs only when its suffix has no `gpu_metrics_` + * upload, the same preference `discoverGpuMetricsArtifacts` applies. + */ +export function pairGpuMetricsArtifacts( + artifacts: readonly ArtifactMeta[], +): GpuMetricsArtifactPair[] { + const byName = new Map(); + for (const artifact of artifacts) { + if (artifact.expired) continue; + const existing = byName.get(artifact.name); + if (!existing || isNewerArtifact(artifact, existing)) byName.set(artifact.name, artifact); + } + const gpuMetricsSuffixes = new Set(); + for (const name of byName.keys()) { + const suffix = gpuMetricsArtifactSuffix(name); + if (suffix && !isPowerAuditArtifact(name)) gpuMetricsSuffixes.add(suffix); + } + const byLogicalBenchmark = new Map(); + for (const gpuMetrics of byName.values()) { + const suffix = gpuMetricsArtifactSuffix(gpuMetrics.name); + if (!suffix) continue; + if (isPowerAuditArtifact(gpuMetrics.name) && gpuMetricsSuffixes.has(suffix)) continue; + const benchmarks = byName.get(`bmk_agentic_${suffix}`) ?? byName.get(`bmk_${suffix}`); + if (!benchmarks) continue; + const logicalName = benchmarks.name.replace(RUNNER_SUFFIX_RE, ''); + const existing = byLogicalBenchmark.get(logicalName); + if (!existing || isNewerArtifact(benchmarks, existing.benchmarks)) { + byLogicalBenchmark.set(logicalName, { gpuMetrics, benchmarks }); + } + } + return [...byLogicalBenchmark.values()].toSorted((a, b) => + a.gpuMetrics.name.localeCompare(b.gpuMetrics.name), + ); +} + +/** A corrupt unrelated benchmark sibling must not prevent valid pairs from being repaired. */ +export async function collectMissingTelemetryExpectations( + artifacts: readonly ArtifactMeta[], + pairs: readonly GpuMetricsArtifactPair[], + selectedArtifact: string | null, + readRows: ( + artifact: ArtifactMeta, + onUnmapped: (error: string) => void, + ) => Promise, +): Promise<{ + observations: TelemetryObservation[]; + errors: NonNullable; +}> { + const paired = new Set(pairs.map((pair) => pair.benchmarks.name)); + const observations: TelemetryObservation[] = []; + const errors: NonNullable = []; + for (const artifact of dedupeArtifactsByLogicalName(artifacts).values()) { + if (!artifact.name.startsWith('bmk_') || paired.has(artifact.name)) continue; + const suffix = artifact.name.replace(/^bmk_(?:agentic_)?/u, ''); + const artifactNames = [`gpu_metrics_${suffix}`, `power_audit_${suffix}`]; + if (selectedArtifact && !artifactNames.includes(selectedArtifact)) continue; + const recordError = (error: string) => { + errors.push({ benchmarkArtifact: artifact.name, artifactNames, error }); + }; + if (artifact.expired) { + recordError('Benchmark artifact expired; point identities unavailable'); + continue; + } + try { + for (const row of await readRows(artifact, recordError)) + observations.push({ + identity: benchmarkPublicationIdentity(row), + artifactNames, + produced: false, + }); + } catch (error) { + recordError(error instanceof Error ? error.message : String(error)); + } + } + return { observations, errors }; +} + +/** Stored samples outlive artifacts, including superseded attempts. */ +export function findOutdatedGpuMetricSeries( + sql: Sql, + flags: { + run: number | null; + attempt: number | null; + artifact: string | null; + fromRun: number | null; + since: string | null; + }, + limit: number | null, +) { + return sql` + select s.id, wr.github_run_id, wr.run_attempt, s.artifact_name, s.file_name + from gpu_metric_series s join workflow_runs wr on wr.id = s.workflow_run_id + where s.stats_version <> ${GPU_STATS_VERSION} + and (${flags.run}::bigint is null or wr.github_run_id = ${flags.run}) + and (${flags.attempt}::integer is null or wr.run_attempt = ${flags.attempt}) + and (${flags.artifact}::text is null or s.artifact_name = ${flags.artifact}) + and (${flags.fromRun}::bigint is null or wr.github_run_id >= ${flags.fromRun}) + and (${flags.since}::date is null or wr.date >= ${flags.since}::date) + order by wr.github_run_id, wr.run_attempt, s.id + limit ${limit}::integer + `; +} diff --git a/packages/db/src/lib/telemetry-purge.ts b/packages/db/src/lib/telemetry-purge.ts new file mode 100644 index 000000000..13b2b76dd --- /dev/null +++ b/packages/db/src/lib/telemetry-purge.ts @@ -0,0 +1,100 @@ +/** + * Explicit deletion of PowerX telemetry during a purge. + * + * Migration 016 declares `on delete cascade` from `gpu_metric_series.workflow_run_id` + * to `workflow_runs` and from `benchmark_result_gpu_metrics.benchmark_result_id` to + * `benchmark_results`, so a purge already removed telemetry — silently, with no count + * in the preview and no line in the log. That is the wrong default for this data: + * past GitHub's 90-day artifact retention the stored samples are the only copy, which + * is the reason migration 016 exists at all. These helpers make the deletion explicit + * so the operator sees the cost before confirming and again in the transcript. + * + * The two levels differ on purpose: + * - A whole-run purge owns the series and deletes them (`deleteRunTelemetry`). + * - A point purge only drops the point→series links (`unlinkPointTelemetry`). The + * series stays attached to its workflow_run because `/api/gpu-metrics?runId=` + * reads series by run, not through the links, and other points of the same run + * may still reference it. + */ + +import type { Sql } from '../etl/db-utils.js'; + +export interface TelemetryPurgeCounts { + /** Rows in `gpu_metric_series`. */ + series: number; + /** Sum of `gpu_metric_series.sample_count` across those series. */ + samples: number; +} + +export const NO_TELEMETRY: TelemetryPurgeCounts = { series: 0, samples: 0 }; + +/** `count`/`sum` come back as strings over the wire; `sum` is null on an empty set. */ +function toCounts(row: { n: unknown; samples: unknown } | undefined): TelemetryPurgeCounts { + if (!row) return NO_TELEMETRY; + return { series: Number(row.n ?? 0), samples: Number(row.samples ?? 0) }; +} + +/** + * Telemetry a whole-run purge would destroy. Read-only, for the preview line. + * Sums the stored `sample_count` instead of counting `gpu_metric_samples` so the + * preview stays cheap against a table holding tens of millions of rows. + */ +export async function countRunTelemetry( + sql: Sql, + workflowRunIds: readonly number[], +): Promise { + if (workflowRunIds.length === 0) return NO_TELEMETRY; + const [row] = await sql` + SELECT count(*)::int AS n, coalesce(sum(sample_count), 0)::bigint AS samples + FROM gpu_metric_series + WHERE workflow_run_id = ANY(${[...workflowRunIds]}) + `; + return toCounts(row as { n: unknown; samples: unknown } | undefined); +} + +/** + * Delete the telemetry owned by these workflow_runs and report what went, so the + * caller can log it. Samples and per-GPU stats follow by cascade from the series. + * Call this before deleting the `workflow_runs` rows themselves. + */ +export async function deleteRunTelemetry( + sql: Sql, + workflowRunIds: readonly number[], +): Promise { + if (workflowRunIds.length === 0) return NO_TELEMETRY; + const [row] = await sql` + WITH deleted AS ( + DELETE FROM gpu_metric_series + WHERE workflow_run_id = ANY(${[...workflowRunIds]}) + RETURNING sample_count + ) + SELECT count(*)::int AS n, coalesce(sum(sample_count), 0)::bigint AS samples FROM deleted + `; + return toCounts(row as { n: unknown; samples: unknown } | undefined); +} + +/** + * Drop the point→series links for purged benchmark points and report how many went. + * Deliberately leaves `gpu_metric_series` in place: it belongs to the workflow_run, + * which is not being purged here. + */ +export async function unlinkPointTelemetry( + sql: Sql, + benchmarkResultIds: readonly number[], +): Promise { + if (benchmarkResultIds.length === 0) return 0; + const [row] = await sql` + WITH deleted AS ( + DELETE FROM benchmark_result_gpu_metrics + WHERE benchmark_result_id = ANY(${[...benchmarkResultIds]}) + RETURNING series_id + ) + SELECT count(*)::int AS n FROM deleted + `; + return Number((row as { n: unknown } | undefined)?.n ?? 0); +} + +/** One-line summary for preview and transcript output. */ +export function describeTelemetry(counts: TelemetryPurgeCounts): string { + return `${counts.series} gpu_metric_series (${counts.samples.toLocaleString('en-US')} samples)`; +} diff --git a/packages/db/src/preflight-power-publication.ts b/packages/db/src/preflight-power-publication.ts new file mode 100644 index 000000000..031c07170 --- /dev/null +++ b/packages/db/src/preflight-power-publication.ts @@ -0,0 +1,36 @@ +import { verifyRequiredPowerArtifacts } from './etl/required-power-publication'; + +// Deliberately artifact-only: this entry point opens no DB connection and writes no files. +const [root, runId, runAttempt, headSha, option] = process.argv.slice(2); +if ( + !root || + !runId || + !runAttempt || + !/^\d+$/u.test(runId) || + !/^\d+$/u.test(runAttempt) || + !Number.isSafeInteger(Number(runId)) || + !Number.isSafeInteger(Number(runAttempt)) || + Number(runId) <= 0 || + Number(runAttempt) <= 0 || + !headSha || + !/^[a-f0-9]{40}$/u.test(headSha) || + (option !== undefined && option !== '--optional-power') || + process.argv.length > 7 +) + throw new Error( + 'Usage: preflight-power-publication.ts [--optional-power]', + ); + +const points = verifyRequiredPowerArtifacts( + root, + { runId: Number(runId), runAttempt: Number(runAttempt), headSha }, + option !== '--optional-power', +); +console.log( + JSON.stringify({ + status: 'validated_artifacts', + requiredPoints: points.length, + databaseWrites: 0, + curvePreservation: 'not_checked_requires_published_state', + }), +); diff --git a/packages/db/src/prepare-ci-artifacts.ts b/packages/db/src/prepare-ci-artifacts.ts index fbf834d70..f0c35780e 100644 --- a/packages/db/src/prepare-ci-artifacts.ts +++ b/packages/db/src/prepare-ci-artifacts.ts @@ -6,6 +6,7 @@ import path from 'node:path'; import { buildArtifactPlan } from './lib/ci-artifact-preparation.js'; import { downloadArtifact, listRunArtifacts, type ArtifactMeta } from './lib/github-artifacts.js'; +import { verifyRequiredPowerArtifacts } from './etl/required-power-publication.js'; const DEFAULT_REPO = 'SemiAnalysisAI/InferenceX'; @@ -148,6 +149,16 @@ function main(): void { console.log(`Downloading artifact: ${artifact.name}`); downloadWithRetries(artifact, artifactsPath); } + // This command precedes migrations: malformed required input must not reach any DB write. + verifyRequiredPowerArtifacts( + artifactsPath, + { + runId: Number(sourceRunId), + runAttempt: sourceMetadata.run_attempt ?? 1, + headSha: sourceMetadata.head_sha ?? null, + }, + process.env.INGEST_REQUIRE_POWER === 'true', + ); if (plan.reused) { writeReuseMetadata(artifactsPath, sourceRunId, mergeRunId, sourceMetadata, mergeMetadata); } diff --git a/packages/db/src/queries/gpu-metrics-revision.ts b/packages/db/src/queries/gpu-metrics-revision.ts new file mode 100644 index 000000000..fb7364b38 --- /dev/null +++ b/packages/db/src/queries/gpu-metrics-revision.ts @@ -0,0 +1,27 @@ +import { GPU_STATS_VERSION } from '../lib/gpu-metric-stats'; + +import type { DbClient } from '../connection'; + +/** Re-ingest and shared-link changes must bypass older point payloads. */ +export async function getGpuMetricsPointRevision( + sql: DbClient, + benchmarkResultId: number, +): Promise { + const [row] = await sql` + select md5(jsonb_agg(jsonb_build_array( + s.id, s.ingested_at, s.csv_sha256, s.sample_count, s.sidecars, + coalesce((to_jsonb(s)->>'stats_version')::integer, 0), + ( + select jsonb_agg(jsonb_build_array(l.benchmark_result_id, b.power_audit) + order by l.benchmark_result_id) + from benchmark_result_gpu_metrics l + join benchmark_results b on b.id = l.benchmark_result_id + where l.series_id = s.id + ) + ) order by s.id)::text) as revision + from benchmark_result_gpu_metrics link + join gpu_metric_series s on s.id = link.series_id + where link.benchmark_result_id = ${benchmarkResultId} + `; + return typeof row?.revision === 'string' ? `${GPU_STATS_VERSION}:${row.revision}` : null; +} diff --git a/packages/db/src/queries/gpu-metrics-timeline.test.ts b/packages/db/src/queries/gpu-metrics-timeline.test.ts new file mode 100644 index 000000000..2283453ce --- /dev/null +++ b/packages/db/src/queries/gpu-metrics-timeline.test.ts @@ -0,0 +1,288 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { PGlite } from '@electric-sql/pglite'; +import type postgres from 'postgres'; +import { afterAll, beforeAll, beforeEach, describe, expect, it } from 'vitest'; + +import { cutPowerAuditBundle } from '../../../app/src/components/gpu-power/power-audit-bundle'; +import { storedPowerSeries } from '../../../app/src/components/gpu-power/stored-power-series'; +import type { DbClient } from '../connection'; +import { ingestGpuMetricsArtifact } from '../etl/gpu-metrics-ingest'; +import { getGpuMetricsForPoint, getGpuMetricsForRun } from './gpu-metrics'; + +let db: PGlite; +let sql: postgres.Sql; +let readSql: DbClient; +let root: string; +const RUN = 34557177019; +const NAME = 'power_audit_qwen3.5_8k1k_fp8_dynamo-sglang_b200-slurm_0'; +const SOURCE = 'power_validation_qwen3.5_8k1k_fp8_dynamo-sglang_b200-slurm_0_conc32.json'; +const START = 1789194365; +const CSV = [ + 'schema_version,timestamp_unix,scrape_seq,hostname,gpu_index,gpu_uuid,power_w', + `1,${START - 61},0,host-a,0,GPU-a0,1`, // outside the existing 60 s pad + `1,${START - 60},1,host-a,0,GPU-a0,0`, // genuine zero must survive + `1,${START + 0.1},2,host-b,0,GPU-b0,450`, + `1,${START + 0.1},2,host-a,0,GPU-a0,300`, + `1,${START + 0.1},2,host-a,0,GPU-a0,300`, // identical duplicate flush + `1,${START + 0.1},2,host-a,0,GPU-a0,999`, // conflicting duplicate: first wins + `1,${START + 0.8},3,host-a,0,GPU-a0,500`, // same second: mean 400 W + `1,${START + 1.1},4,host-a,0,GPU-a0,550`, // host-b dropout: null bucket + `1,${START + 61},5,host-b,0,GPU-b0,90`, // inclusive padded boundary + `1,${START + 62},6,host-b,0,GPU-b0,1`, +].join('\n'); +const MANIFEST = { + producer: 'srt-slurm.dcgm-power', + expected_devices: [ + { hostname: 'host-a', gpu_index: 0, assignments: [{ worker_role: 'prefill' }] }, + { hostname: 'host-b', gpu_index: 0, assignments: [{ worker_role: 'decode' }] }, + ], +}; +const VALIDATION = { + selected_window: { start_time_unix: START, end_time_unix: START + 1 }, + benchmark_window: { start_time_unix: START - 100, end_time_unix: START + 100 }, +}; + +function stamp(seconds: number): string { + return new Date((seconds + 7200) * 1000) + .toISOString() + .replaceAll('-', '/') + .replace('T', ' ') + .replace('Z', ''); +} + +function client(database: Pick) { + return Object.assign( + async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await database.query>(query, values); + return result.rows; + }, + { json: JSON.stringify, array: (value: unknown) => value }, + ); +} + +function writeArtifact(name: string, files: ReadonlyMap) { + const artifactDir = path.join(root, name); + for (const [file, contents] of files) { + const pathname = path.join(artifactDir, file); + fs.mkdirSync(path.dirname(pathname), { recursive: true }); + fs.writeFileSync(pathname, contents); + } + return { artifactName: name, artifactDir }; +} + +function bundle(validation: Record = VALIDATION) { + return new Map([ + ['LOGS/power/samples.csv', CSV], + ['LOGS/power/manifest.json', JSON.stringify(MANIFEST)], + [SOURCE, JSON.stringify(validation)], + ]); +} + +function agentxBundle() { + const files = bundle(); + files.delete(SOURCE); + const resultPath = 'agentic/conc_32/agentic_power_concurrency_32.json'; + files.set(`${NAME.slice('power_audit_'.length)}_conc32.json`, JSON.stringify({ conc: 32 })); + files.set( + 'LOGS/agentic/conc_32/power_validation.json', + JSON.stringify({ + ...VALIDATION, + power_valid: false, + selected_window: { + ...VALIDATION.selected_window, + concurrency: 32, + result_path: resultPath, + window_file: 'windows/agentic_power_concurrency_32.json', + }, + }), + ); + files.set( + 'LOGS/power/windows/agentic_power_concurrency_32.json', + JSON.stringify({ + concurrency: 32, + result_path: resultPath, + benchmark_start_time_unix: START, + benchmark_end_time_unix: START + 1, + }), + ); + return files; +} + +beforeAll(async () => { + root = fs.mkdtempSync(path.join(os.tmpdir(), 'powerx-timeline-')); + db = await PGlite.create(); + for (const name of [ + '001_initial_schema.sql', + '016_gpu_metrics.sql', + '017_gpu_metric_stats_version.sql', + ]) { + await db.exec(fs.readFileSync(new URL(`../../migrations/${name}`, import.meta.url), 'utf8')); + } + await db.exec('ALTER TABLE benchmark_results ADD COLUMN power_audit jsonb'); + readSql = client(db); + sql = Object.assign(client(db), { + begin: (fn: (tx: postgres.Sql) => Promise) => + db.transaction((tx) => fn(client(tx) as unknown as postgres.Sql)), + }) as unknown as postgres.Sql; +}, 20_000); + +afterAll(async () => { + await db?.close(); + fs.rmSync(root, { recursive: true, force: true }); +}); + +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, status, created_at, date) + VALUES (1, ${RUN}, 1, 'Run Sweep', 'completed', '2026-09-12', '2026-09-12'); + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'qwen3.5', 'b200', 'sglang', 'fp8', 'none', false, 1, 1, 1, 1); + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, + isl, osl, conc, metrics) + VALUES (10, 1, 1, 'single_turn', '2026-09-12', 8192, 1024, 32, '{}'), + (11, 1, 1, 'single_turn', '2026-09-12', 8192, 1024, 64, '{}');`); +}); + +describe('artifact → ingest → stored Timeline', () => { + it.each(['explicit', 'wrong-run', 'wrong-id', 'wrong-conc', 'ambiguous'])( + 'does not infer or overwrite point provenance for %s identity', + async (scenario) => { + await db.exec("UPDATE benchmark_results SET benchmark_type = 'agentic_traces'"); + if (scenario === 'explicit') + await db.exec( + `UPDATE benchmark_results SET power_audit = '{"source":"existing.json"}' WHERE id = 10`, + ); + if (scenario === 'wrong-run') { + await db.exec(`INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, status, created_at, date) + VALUES (2, 1, 1, 'Other run', 'completed', '2026-09-12', '2026-09-12'); + UPDATE benchmark_results SET workflow_run_id = 2 WHERE id = 10`); + } + if (scenario === 'wrong-conc') + await db.exec('UPDATE benchmark_results SET conc = 31 WHERE id = 10'); + if (scenario === 'ambiguous') + await db.exec('UPDATE benchmark_results SET conc = 32, isl = 1 WHERE id = 11'); + const before = await db.query('SELECT id, power_audit FROM benchmark_results ORDER BY id'); + await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: writeArtifact(NAME, agentxBundle()), + benchmarkResultIds: scenario === 'wrong-id' ? [11] : [10, 11], + }); + const after = await db.query('SELECT id, power_audit FROM benchmark_results ORDER BY id'); + expect(after.rows).toEqual(before.rows); + }, + ); + + it('matches artifact windows, host/GPU identity, deduplicated mean watts and gaps for a multinode bundle', async () => { + const files = bundle(); + const artifact = writeArtifact(NAME, files); + await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact, + benchmarkResultIds: [10, 11], + }); + // Remove the actual artifact before querying: only the DB survives. + fs.rmSync(artifact.artifactDir, { recursive: true }); + const stored = await getGpuMetricsForRun(readSql, RUN); + const actual = storedPowerSeries(stored!.series); + expect(actual).toEqual(cutPowerAuditBundle(NAME, files)); + expect(actual[0]).toMatchObject({ + artifact: NAME, + source: SOURCE, + startMs: (START - 60) * 1000, + t: [0, 60, 61, 121], + power: [ + [0, 400, 550, null], + [null, 450, null, 90], + ], + devices: [ + { id: 'host-a/GPU-a0', role: 'prefill' }, + { id: 'host-b/GPU-b0', role: 'decode' }, + ], + }); + const point = await getGpuMetricsForPoint(readSql, 10); + expect(point!.series.map((series) => series.benchmarkResultIds)).toEqual([ + [10, 11], + [10, 11], + ]); + expect(storedPowerSeries(point!.series)).toEqual(actual); + }); + + it('retains selected-window precedence and validation role overrides across ingestion', async () => { + const files = bundle({ + ...VALIDATION, + per_gpu_role: { 'host-a/GPU-a0': 'decode', 'host-b/GPU-b0': 'prefill' }, + }); + await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: writeArtifact(NAME, files), + benchmarkResultIds: [10], + }); + const stored = await getGpuMetricsForRun(readSql, RUN); + const actual = storedPowerSeries(stored!.series); + expect(actual).toEqual(cutPowerAuditBundle(NAME, files)); + expect(actual[0].devices?.[0]).toEqual({ id: 'host-b/GPU-b0', role: 'prefill' }); + }); + + it('keeps SMI telemetry embedded in a power_audit bundle readable with the same source/window cuts', async () => { + const files = bundle(); + files.set( + 'gpu_metrics.csv', + [ + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]', + `${stamp(START - 61)}, 0, 1 W, 60, 1000 MHz, 2000 MHz, 80 %, 70 %`, + `${stamp(START)}, 0, 300 W, 60, 1000 MHz, 2000 MHz, 80 %, 70 %`, + `${stamp(START)}, 1, 500 W, 60, 1000 MHz, 2000 MHz, 80 %, 70 %`, + `${stamp(START + 1)}, 0, 700 W, 60, 1000 MHz, 2000 MHz, 80 %, 70 %`, + ].join('\n'), + ); + files.set('gpu_metrics_identity.csv', 'index, uuid\n0, GPU-a\n1, GPU-b'); + files.set( + 'benchmark_gpu_metrics_context.json', + JSON.stringify({ timestamp_timezone: '+02:00' }), + ); + const artifact = writeArtifact(NAME, files); + await ingestGpuMetricsArtifact(sql, { workflowRunId: 1, artifact, benchmarkResultIds: [10] }); + fs.rmSync(artifact.artifactDir, { recursive: true }); + const stored = await getGpuMetricsForRun(readSql, RUN); + expect(stored!.series).toHaveLength(1); + expect(stored!.series[0].fileName).toBe('gpu_metrics.csv'); + const actual = storedPowerSeries(stored!.series); + expect(actual).toEqual(cutPowerAuditBundle(NAME, files)); + expect(actual).toMatchObject([ + { + source: SOURCE, + startMs: START * 1000, + gpus: [0, 1], + power: [ + [300, 700], + [500, null], + ], + devices: [{ id: '0' }, { id: '1' }], + }, + ]); + }); + + it('rejects incomplete per-host commits using the artifact inventory and legacy manifest', async () => { + await ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: writeArtifact(NAME, bundle()), + benchmarkResultIds: [10], + }); + await sql`DELETE FROM gpu_metric_series WHERE file_name = 'LOGS/power/samples.csv#host-b'`; + const stored = await getGpuMetricsForRun(readSql, RUN); + expect(stored!.series).toHaveLength(1); + expect(() => storedPowerSeries(stored!.series)).toThrow( + 'stored file/sample coverage is incomplete', + ); + await sql`UPDATE gpu_metric_series SET sidecars = sidecars - 'seriesInventory'`; + const legacy = await getGpuMetricsForRun(readSql, RUN); + expect(() => storedPowerSeries(legacy!.series)).toThrow( + 'expected device host-b GPU 0 is not stored', + ); + }); +}); diff --git a/packages/db/src/queries/gpu-metrics.reingest-consistency.test.ts b/packages/db/src/queries/gpu-metrics.reingest-consistency.test.ts new file mode 100644 index 000000000..7cba11008 --- /dev/null +++ b/packages/db/src/queries/gpu-metrics.reingest-consistency.test.ts @@ -0,0 +1,236 @@ +/** + * Re-ingest consistency of the PowerX telemetry read path. + * + * Property under test: a reader of one benchmark point never observes samples + * and per-GPU statistics that come from different versions of the same series. + * + * `upsertGpuMetricSeries` replaces samples and stats inside one `sql.begin` + * transaction, so the write side is atomic. `getGpuMetricsForPoint` -> + * `loadSeriesDetails`, however, issues separate autocommit statements (series + * rows, then `gpu_metric_gpu_stats`, then `gpu_metric_samples`). In production + * the DbClient is either neon() HTTP (one implicit transaction per call) or a + * postgres.js pool used without a transaction, so a re-ingest can commit between + * the stats and the samples statement. The reader re-checks the series version + * key afterwards and restarts the read when it moved. + * + * PGlite is single-connection, so the interleaving is simulated at statement + * granularity: after the stats statement returns, a full re-ingest transaction + * runs and commits before the samples statement executes. Under READ COMMITTED + * this is observationally identical to a second connection committing between + * the two autocommit statements. + */ + +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +import { PGlite } from '@electric-sql/pglite'; +import type postgres from 'postgres'; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from 'vitest'; + +import type { DbClient } from '../connection'; +import { ingestGpuMetricsArtifact, refreshGpuMetricStats } from '../etl/gpu-metrics-ingest'; +import { getGpuMetricsForPoint, type GpuMetricSeries } from './gpu-metrics'; + +type Sql = postgres.Sql; +let db: PGlite; +/** Writer handle: the ingest code calls `sql.begin`, `sql.array`, `sql.json`. */ +let sql: Sql; +/** Reader handle: plain autocommit statements, as the app's DbClient issues them. */ +let reader: DbClient; +const roots: string[] = []; + +function queryClient(database: Pick) { + const client = async (strings: TemplateStringsArray, ...values: unknown[]) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await database.query>(query, values); + return result.rows; + }; + return Object.assign(client, { + json: JSON.stringify, + array: (value: unknown) => value, + }); +} + +beforeAll(async () => { + db = await PGlite.create(); + for (const name of [ + '001_initial_schema.sql', + '016_gpu_metrics.sql', + '017_gpu_metric_stats_version.sql', + ]) { + await db.exec(fs.readFileSync(new URL(`../../migrations/${name}`, import.meta.url), 'utf8')); + } + // The point reader aggregates benchmark_results.power_audit (migration 015). + await db.exec('ALTER TABLE benchmark_results ADD COLUMN power_audit jsonb'); + sql = Object.assign(queryClient(db), { + begin: (fn: (tx: Sql) => Promise) => + db.transaction((tx) => fn(queryClient(tx) as unknown as Sql)), + }) as unknown as Sql; + reader = queryClient(db) as DbClient; +}, 20_000); + +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, status, conclusion, created_at, date) + VALUES (1, 34557177019, 1, 'Run Sweep', 'completed', 'success', '2026-09-11', '2026-09-11'); + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, 4, 4, 4, 4); + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, isl, osl, conc, metrics) + VALUES (10, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, '{}'), + (11, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 64, '{}');`); +}); + +afterEach(() => { + for (const root of roots.splice(0)) fs.rmSync(root, { recursive: true, force: true }); +}); + +afterAll(async () => { + await db?.close(); +}); + +/** Version 1: 2 GPUs x 2 distinct timestamps (+ duplicate flush row) = 4 unique samples. */ +const NVIDIA_CSV_V1 = [ + 'timestamp, index, power.draw [W], temperature.gpu, clocks.current.sm [MHz], clocks.current.memory [MHz], utilization.gpu [%], utilization.memory [%]', + '2026/09/11 04:19:41.982, 0, 187.80 W, 33, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:41.986, 1, 190.96 W, 39, 120 MHz, 3996 MHz, 0 %, 0 %', + '2026/09/11 04:19:42.990, 0, 912.10 W, 61, 1965 MHz, 3996 MHz, 98 %, 74 %', + '2026/09/11 04:19:42.994, 1, 905.30 W, 62, 1965 MHz, 3996 MHz, 97 %, 73 %', + // Duplicate flush of the last sample, as emitted when the monitor stops. + '2026/09/11 04:19:42.994, 1, 905.30 W, 62, 1965 MHz, 3996 MHz, 97 %, 73 %', +].join('\n'); + +/** + * Version 1 + `extraScrapes` further scrapes per GPU, each at a distinct + * timestamp, so every version has a distinct CSV digest and sample count. + */ +function csvVersion(extraScrapes: number): string { + const lines = [NVIDIA_CSV_V1]; + for (let i = 0; i < extraScrapes; i += 1) { + const second = String(43 + i).padStart(2, '0'); + lines.push( + `2026/09/11 04:19:${second}.990, 0, 700.00 W, 60, 1900 MHz, 3996 MHz, 90 %, 70 %`, + `2026/09/11 04:19:${second}.994, 1, 702.50 W, 61, 1900 MHz, 3996 MHz, 91 %, 71 %`, + ); + } + return lines.join('\n'); +} + +/** Version 2: the same artifact re-uploaded with one more scrape per GPU = 6 unique samples. */ +const NVIDIA_CSV_V2 = csvVersion(1); + +const V1_UNIQUE_SAMPLES = 4; +const V2_UNIQUE_SAMPLES = 6; + +function writeArtifact(csv: string, contextZone = 'UTC') { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gpu-metrics-reingest-')); + roots.push(root); + const artifactName = 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0'; + const artifactDir = path.join(root, artifactName); + fs.mkdirSync(artifactDir); + fs.writeFileSync(path.join(artifactDir, 'gpu_metrics.csv'), csv); + fs.writeFileSync( + path.join(artifactDir, 'gpu_metrics_context.json'), + JSON.stringify({ timestamp_timezone: contextZone }), + ); + fs.writeFileSync( + path.join(artifactDir, 'gpu_metrics_identity.csv'), + 'index, uuid, name\n0, GPU-a, NVIDIA B200\n1, GPU-b, NVIDIA B200\n', + ); + return { artifactName, artifactDir }; +} + +function ingest(csv: string) { + return ingestGpuMetricsArtifact(sql, { + workflowRunId: 1, + artifact: writeArtifact(csv), + benchmarkResultIds: [10], + }); +} + +interface InterleavedReader { + client: DbClient; + /** How many times the reader issued its first statement (the point's series rows). */ + seriesReads: () => number; +} + +/** + * A DbClient that forwards every statement to `base`, but after each + * `gpu_metric_gpu_stats` statement has returned (and before the caller gets + * the rows back), runs `between(statsReadIndex)` to completion. With a + * committed re-ingest in `between`, this is the schedule: reader(series), + * reader(stats), WRITER COMMITS, reader(samples), reader(version check). + */ +function interleaveAfterStats( + base: DbClient, + between: (statsReadIndex: number) => Promise, +): InterleavedReader { + let statsReads = 0; + let seriesReads = 0; + const client: DbClient = async (strings, ...values) => { + const text = strings.join(''); + if (/from benchmark_result_gpu_metrics link/.test(text)) seriesReads += 1; + const rows = await base(strings, ...values); + if (/from gpu_metric_gpu_stats/.test(text)) { + statsReads += 1; + await between(statsReads); + } + return rows; + }; + return { client, seriesReads: () => seriesReads }; +} + +/** + * The consistency property: every statistic row was computed over exactly the + * sample rows returned in the same payload, and the series header agrees. + */ +function expectCoherent(series: GpuMetricSeries) { + for (const stat of series.stats) { + const rowsForGpu = series.data.filter((row) => row.index === stat.gpuIndex).length; + expect(stat.count, `stats.count for gpu ${stat.gpuIndex}/${stat.metric}`).toBe(rowsForGpu); + } + expect(series.sampleCount, 'series.sampleCount vs data.length').toBe(series.data.length); +} + +describe('getGpuMetricsForPoint under telemetry re-ingest', () => { + it('a re-ingest committed between the stats and samples statements yields one version', async () => { + const first = await ingest(NVIDIA_CSV_V1); + expect(first.samplesInserted).toBe(V1_UNIQUE_SAMPLES); + + let reingest: Awaited> | undefined; + const interleaved = interleaveAfterStats(reader, async (statsReadIndex) => { + if (statsReadIndex === 1) reingest = await ingest(NVIDIA_CSV_V2); + }); + + const payload = await getGpuMetricsForPoint(interleaved.client, 10); + + // The writer really did replace the series inside its own transaction. + expect(reingest?.samplesInserted).toBe(V2_UNIQUE_SAMPLES); + expect(reingest?.seriesIds).toEqual(first.seriesIds); + expect(payload?.series).toHaveLength(1); + // The version key moved, so the reader restarted from the series statement. + expect(interleaved.seriesReads()).toBe(2); + + for (const series of payload!.series) { + expect(series.sampleCount).toBe(V2_UNIQUE_SAMPLES); + expect(series.data).toHaveLength(V2_UNIQUE_SAMPLES); + for (const stat of series.stats) expect(stat.count).toBe(V2_UNIQUE_SAMPLES / 2); + expectCoherent(series); + } + }); +}); + +it('retries when only the digest version changes between statements', async () => { + const first = await ingest(NVIDIA_CSV_V1); + await sql`update gpu_metric_series set stats_version = 0`; + await sql`update gpu_metric_gpu_stats set mean_value = -1`; + const interleaved = interleaveAfterStats(reader, async (index) => { + if (index === 1) await refreshGpuMetricStats(sql, first.seriesIds[0]!); + }); + const payload = await getGpuMetricsForPoint(interleaved.client, 10); + expect(interleaved.seriesReads()).toBe(2); + expect( + payload?.series[0]?.stats.find((s) => s.gpuIndex === 0 && s.metric === 'power_w')?.mean, + ).toBeCloseTo(549.95, 3); +}); diff --git a/packages/db/src/queries/gpu-metrics.test.ts b/packages/db/src/queries/gpu-metrics.test.ts new file mode 100644 index 000000000..54745f11a --- /dev/null +++ b/packages/db/src/queries/gpu-metrics.test.ts @@ -0,0 +1,321 @@ +import fs from 'node:fs'; + +import { PGlite } from '@electric-sql/pglite'; +import { afterAll, beforeAll, beforeEach, describe, expect, it } from 'vitest'; + +import type { DbClient } from '../connection'; +import { getGpuMetricsForPoint, getGpuMetricsForRun, SAMPLE_PAGE_SIZE } from './gpu-metrics'; + +let db: PGlite; +const sql: DbClient = async (strings, ...values) => { + const query = strings.reduce((text, part, i) => text + (i ? `$${i}` : '') + part, ''); + const result = await db.query>(query, values); + return result.rows; +}; + +const WITH_SERIES = 34557177019; +const RETRIED = 34557177021; + +beforeAll(async () => { + db = await PGlite.create(); + for (const name of ['001_initial_schema.sql', '016_gpu_metrics.sql']) { + await db.exec(fs.readFileSync(new URL(`../../migrations/${name}`, import.meta.url), 'utf8')); + } + await db.exec('ALTER TABLE benchmark_results ADD COLUMN power_audit jsonb'); +}, 20_000); + +afterAll(async () => { + await db?.close(); +}); + +/** + * Run 1 holds a two-node artifact (one CSV per serving node) whose first CSV is + * shared by points 10 and 11; runs 3 and 4 are two attempts of one GitHub run. + * Point 12 is deliberately left unlinked. + */ +beforeEach(async () => { + await db.exec(`TRUNCATE workflow_runs, configs RESTART IDENTITY CASCADE; + INSERT INTO workflow_runs (id, github_run_id, run_attempt, name, status, conclusion, + head_branch, head_sha, html_url, created_at, date) + VALUES (1, ${WITH_SERIES}, 1, 'Run Sweep', 'completed', 'success', 'main', 'abc123', + 'https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${WITH_SERIES}', + '2026-09-11T04:19:00Z', '2026-09-11'), + (3, ${RETRIED}, 1, 'Run Sweep', 'completed', 'failure', null, null, null, + '2026-09-13T04:19:00Z', '2026-09-13'), + (4, ${RETRIED}, 2, 'Run Sweep', 'completed', 'success', null, null, null, + '2026-09-13T06:19:00Z', '2026-09-13'); + + INSERT INTO configs (id, model, hardware, framework, precision, spec_method, disagg, + is_multinode, prefill_tp, decode_tp, num_prefill_gpu, num_decode_gpu) + VALUES (1, 'dsr1', 'b200', 'sglang', 'fp4', 'none', false, true, 8, 8, 8, 8); + + INSERT INTO benchmark_results (id, workflow_run_id, config_id, benchmark_type, date, + isl, osl, conc, metrics) + VALUES (10, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 32, '{}'), + (11, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 64, '{}'), + (12, 1, 1, 'single_turn', '2026-09-11', 8192, 1024, 128, '{}'); + + INSERT INTO gpu_metric_series (id, workflow_run_id, artifact_name, config_key, file_name, + vendor, csv_sha256, sample_interval_s, sample_count, gpu_count, started_at, ended_at, + sidecars) + VALUES (100, 1, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', 'node0/gpu_metrics.csv', 'nvidia', 'sha-node0', + 1, 3, 2, '2026-09-11T04:19:41Z', '2026-09-11T04:19:43Z', + '{"context": {"timestamp_timezone": "UTC"}}'), + (101, 1, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', 'node1/gpu_metrics.csv', 'nvidia', 'sha-node1', + 1, 1, 1, '2026-09-11T04:19:41Z', '2026-09-11T04:19:42Z', '{}'), + (102, 4, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_1', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_1', 'gpu_metrics.csv', 'amd', 'sha-attempt2', + null, 1, 1, '2026-09-13T06:20:00Z', '2026-09-13T06:20:01Z', '{}'), + (103, 3, 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', 'gpu_metrics.csv', 'nvidia', 'sha-attempt1', + null, 1, 1, '2026-09-13T04:20:00Z', '2026-09-13T04:20:01Z', '{}'); + + INSERT INTO benchmark_result_gpu_metrics (benchmark_result_id, series_id) + VALUES (10, 100), (11, 100), (10, 101); + + INSERT INTO gpu_metric_gpu_stats (series_id, gpu_index, metric, sample_count, + min_value, max_value, mean_value, median_value, p95_value, p99_value, stddev_value) + VALUES (100, 0, 'powerW', 2, 187.5, 912.25, 549.875, 549.875, 912.25, 912.25, 362.375), + (100, 1, 'powerW', 1, 190.5, 190.5, 190.5, 190.5, 190.5, 190.5, 0), + (101, 0, 'powerW', 1, 500.5, 500.5, 500.5, 500.5, 500.5, 500.5, 0), + (102, 0, 'powerW', 1, 700.5, 700.5, 700.5, 700.5, 700.5, 700.5, 0), + (103, 0, 'powerW', 1, 100.5, 100.5, 100.5, 100.5, 100.5, 100.5, 0); + + INSERT INTO gpu_metric_samples (series_id, gpu_index, sampled_at, power_w, temperature_c, + sm_clock_mhz, mem_clock_mhz, gpu_util_pct, mem_util_pct, edge_temp_c, mem_temp_c, + gfx_voltage_mv, soc_voltage_mv, mem_voltage_mv, fclk_mhz, socclk_mhz, mm_activity_pct) + VALUES (100, 0, '2026-09-11T04:19:41Z', 187.5, 33, 120, 3996, 0, 0, + 35.5, 40.5, 750, 800, 1350, 1940, 1100, 12.5), + (100, 1, '2026-09-11T04:19:41Z', 190.5, 39, 120, 3996, 0, 0, + null, null, null, null, null, null, null, null), + -- Collector dropout: the row exists but every metric column is null. + (100, 0, '2026-09-11T04:19:42Z', null, null, null, null, null, null, + null, null, null, null, null, null, null, null), + (101, 0, '2026-09-11T04:19:41Z', 500.5, 50, 1965, 3996, 98, 74, + null, null, null, null, null, null, null, null), + (102, 0, '2026-09-13T06:20:00Z', 700.5, 60, 1965, 3996, 99, 80, + null, null, null, null, null, null, null, null), + (103, 0, '2026-09-13T04:20:00Z', 100.5, 20, 120, 3996, 0, 0, + null, null, null, null, null, null, null, null);`); +}); + +describe('getGpuMetricsForRun', () => { + it('returns the run header, every series, its per-GPU stats, and its samples', async () => { + const payload = await getGpuMetricsForRun(sql, WITH_SERIES); + + expect(payload?.workflowRun).toEqual({ + id: 1, + githubRunId: WITH_SERIES, + runAttempt: 1, + name: 'Run Sweep', + date: '2026-09-11', + htmlUrl: `https://github.com/SemiAnalysisAI/InferenceX/actions/runs/${WITH_SERIES}`, + headBranch: 'main', + headSha: 'abc123', + conclusion: 'success', + status: 'completed', + createdAt: '2026-09-11T04:19:00.000Z', + }); + + expect(payload?.series.map((series) => series.fileName)).toEqual([ + 'node0/gpu_metrics.csv', + 'node1/gpu_metrics.csv', + ]); + + const [node0, node1] = payload!.series; + expect(node0).toMatchObject({ + id: 100, + artifactName: 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + configKey: 'dsr1_8k1k_fp4_sglang_conc32_b200-x_0', + vendor: 'nvidia', + sampleIntervalS: 1, + sampleCount: 3, + gpuCount: 2, + startedAt: '2026-09-11T04:19:41.000Z', + endedAt: '2026-09-11T04:19:43.000Z', + sidecars: { context: { timestamp_timezone: 'UTC' } }, + benchmarkResultIds: [10, 11], + }); + // Pre-migration (unversioned) digests are recomputed read-only; null is not zero. + expect(node0?.stats.filter((stat) => stat.metric === 'power_w')).toEqual([ + { + gpuIndex: 0, + metric: 'power_w', + count: 1, + min: 187.5, + max: 187.5, + mean: 187.5, + median: 187.5, + p95: 187.5, + p99: 187.5, + stddev: 0, + }, + { + gpuIndex: 1, + metric: 'power_w', + count: 1, + min: 190.5, + max: 190.5, + mean: 190.5, + median: 190.5, + p95: 190.5, + p99: 190.5, + stddev: 0, + }, + ]); + expect(node0?.stats).toEqual( + expect.arrayContaining([ + expect.objectContaining({ gpuIndex: 0, metric: 'edge_temp_c', mean: 35.5, count: 1 }), + expect.objectContaining({ gpuIndex: 0, metric: 'gpu_util_pct', mean: 0, count: 1 }), + expect.objectContaining({ gpuIndex: 0, metric: 'mem_voltage_mv', mean: 1350, count: 1 }), + ]), + ); + expect(node0?.data.map((sample) => [sample.timestamp, sample.index, sample.power])).toEqual([ + ['2026-09-11T04:19:41.000Z', 0, 187.5], + ['2026-09-11T04:19:41.000Z', 1, 190.5], + ['2026-09-11T04:19:42.000Z', 0, 0], + ]); + expect(node0?.data[0]).toEqual({ + timestamp: '2026-09-11T04:19:41.000Z', + index: 0, + power: 187.5, + temperature: 33, + smClock: 120, + memClock: 3996, + gpuUtil: 0, + memUtil: 0, + edgeTemp: 35.5, + memTemp: 40.5, + gfxVoltage: 750, + socVoltage: 800, + memVoltage: 1350, + fclk: 1940, + socClk: 1100, + mmActivity: 12.5, + }); + expect(node1?.benchmarkResultIds).toEqual([10]); + }); + + it('reads the latest attempt of a rerun GitHub run id, not the first', async () => { + const payload = await getGpuMetricsForRun(sql, RETRIED); + expect(payload?.workflowRun).toMatchObject({ id: 4, runAttempt: 2, conclusion: 'success' }); + expect(payload?.series.map((series) => ({ id: series.id, vendor: series.vendor }))).toEqual([ + { id: 102, vendor: 'amd' }, + ]); + }); + + it('reports a dropped sample as zero power with every other metric absent', async () => { + const payload = await getGpuMetricsForRun(sql, WITH_SERIES); + const dropped = payload?.series[0]?.data.at(-1); + // Power keeps the `?? 0` default (a null power reading is indistinguishable + // from a genuine 0 W reading once it reaches the chart); the other metrics + // stay absent so a power-only collector never shows up as 0 °C / 0 MHz / 0 %. + expect(dropped).toEqual({ + timestamp: '2026-09-11T04:19:42.000Z', + index: 0, + power: 0, + temperature: undefined, + smClock: undefined, + memClock: undefined, + gpuUtil: undefined, + memUtil: undefined, + edgeTemp: undefined, + memTemp: undefined, + gfxVoltage: undefined, + socVoltage: undefined, + memVoltage: undefined, + fclk: undefined, + socClk: undefined, + mmActivity: undefined, + }); + // Absent rather than zeroed, so consumers can tell "not collected" from + // "collected as zero" for everything except power. + expect(Object.hasOwn(dropped!, 'edgeTemp')).toBe(true); + expect(dropped?.edgeTemp).toBeUndefined(); + }); +}); + +it('keeps healthy hosts readable when an unversioned sibling has incomplete samples', async () => { + await sql`update gpu_metric_series set sample_count = 4 where id = 100`; + const payload = await getGpuMetricsForRun(sql, WITH_SERIES); + expect(payload?.series[0]?.stats).toEqual([]); + expect(payload?.series[0]?.data).toHaveLength(3); + expect(payload?.series[1]?.stats).toEqual( + expect.arrayContaining([expect.objectContaining({ metric: 'power_w', mean: 500.5, count: 1 })]), + ); +}); + +it('scopes prefix and source reads in SQL, including containing power-audit bundles', async () => { + await db.exec(`INSERT INTO gpu_metric_series + (id, workflow_run_id, artifact_name, config_key, file_name, vendor, csv_sha256, + sample_count, gpu_count, started_at, ended_at) + VALUES (110, 1, 'power_audit_dsr1_8k1k_fp4_sglang', 'bundle', 'samples.csv', 'nvidia', 'bundle', + 0, 1, now(), now()), + (111, 1, 'gpu_metrics_dsr1X8k1k_fp4_sglang_conc32_b200-x_0', 'other', 'gpu_metrics.csv', 'nvidia', 'other', + 0, 1, now(), now());`); + const sampleIds: unknown[] = []; + const observed: DbClient = (strings, ...values) => { + if (strings.join('').includes('from gpu_metric_samples')) sampleIds.push(values[0]); + return sql(strings, ...values); + }; + const payload = await getGpuMetricsForRun(observed, WITH_SERIES, { + prefix: 'dsr1_8k1k_fp4_sglang_conc32', + sourceResults: ['dsr1_8k1k_fp4_sglang_conc32_b200-x_0'], + }); + expect(payload?.series.map((series) => series.id)).toEqual([100, 101, 110]); + expect(sampleIds).toEqual([[100, 101, 110]]); + expect(await getGpuMetricsForRun(sql, WITH_SERIES, { sourceResults: ['missing'] })).toBeNull(); +}); + +it('reads only the selected view host while retaining every exact explorer label', async () => { + const prefix = 'gpu_metrics_dsr1_8k1k_fp4_sglang_conc32_b200-x_0'; + const names = [`${prefix}/node0/gpu_metrics.csv`, `${prefix}/node1/gpu_metrics.csv`]; + const sampleIds: unknown[] = []; + const observed: DbClient = (strings, ...values) => { + if (strings.join('').includes('from gpu_metric_samples')) sampleIds.push(values[0]); + return sql(strings, ...values); + }; + const selected = await getGpuMetricsForRun(observed, WITH_SERIES, { artifact: names[1] }); + expect(selected?.artifactNames).toEqual(names); + expect(selected?.series.map((series) => series.id)).toEqual([101]); + expect(sampleIds).toEqual([[101]]); + const first = await getGpuMetricsForRun(sql, WITH_SERIES, { artifact: null }); + expect(first?.series[0]?.id).toBe(100); + const missing = await getGpuMetricsForRun(observed, WITH_SERIES, { artifact: 'not-an-artifact' }); + expect(missing).toMatchObject({ artifactNames: names, series: [] }); + expect(sampleIds).toHaveLength(1); +}); + +it('pages a single large series without dropping microseconds and retries changes between pages', async () => { + await db.exec(`DELETE FROM gpu_metric_samples WHERE series_id = 100; + INSERT INTO gpu_metric_samples (series_id, gpu_index, sampled_at, power_w) + SELECT 100, 0, '2026-09-11T04:19:41Z'::timestamptz + n * interval '1 microsecond', n + FROM generate_series(1, ${SAMPLE_PAGE_SIZE + 2}) n; + UPDATE gpu_metric_series SET sample_count = ${SAMPLE_PAGE_SIZE + 2} WHERE id = 100;`); + let pages = 0; + const observed: DbClient = async (strings, ...values) => { + const rows = await sql(strings, ...values); + if (strings.join('').includes('from gpu_metric_samples')) { + // Model the bounded HTTP transport; the old whole-series query fails here. + expect(rows.length).toBeLessThanOrEqual(SAMPLE_PAGE_SIZE); + pages++; + if (pages === 1) { + await db.exec(`UPDATE gpu_metric_samples SET power_w = power_w + 20000 WHERE series_id = 100; + UPDATE gpu_metric_series SET ingested_at = ingested_at + interval '1 second' WHERE id = 100;`); + } + } + return rows; + }; + const payload = await getGpuMetricsForPoint(observed, 11); + expect(pages).toBe(4); + const data = payload!.series[0]!.data; + expect(data).toHaveLength(SAMPLE_PAGE_SIZE + 2); + expect(data.map((row) => row.power)).toEqual( + Array.from({ length: SAMPLE_PAGE_SIZE + 2 }, (_, i) => i + 20001), + ); + expect(payload!.series[0]!.stats.find((stat) => stat.metric === 'power_w')?.count).toBe( + SAMPLE_PAGE_SIZE + 2, + ); +}); diff --git a/packages/db/src/queries/gpu-metrics.ts b/packages/db/src/queries/gpu-metrics.ts new file mode 100644 index 000000000..cc4767cb6 --- /dev/null +++ b/packages/db/src/queries/gpu-metrics.ts @@ -0,0 +1,515 @@ +/** + * Read side of the PowerX telemetry digest (migration 016). + * + * Two entry points: everything recorded during one GitHub Actions run (the + * PowerX explorer keyed by run ID), and the series linked to one benchmark + * point (the per-point detail tab). Samples are returned as flat rows with + * ISO timestamps so the existing D3 charts consume them unchanged. + */ + +import { + GPU_STATS_VERSION, + computeStoredGpuMetricStats, + statMetricColumn, + type StoredGpuMetricSample, +} from '../lib/gpu-metric-stats'; + +import type { DbClient } from '../connection.js'; + +export interface GpuMetricSampleRow { + timestamp: string; + index: number; + power: number; + /** + * Absent when the collector did not sample the metric (the multinode DCGM + * power bundle scrapes power only), so readers can tell "not collected" + * from a genuine zero reading. + */ + temperature?: number; + smClock?: number; + memClock?: number; + gpuUtil?: number; + memUtil?: number; + edgeTemp?: number; + memTemp?: number; + gfxVoltage?: number; + socVoltage?: number; + memVoltage?: number; + fclk?: number; + socClk?: number; + mmActivity?: number; +} + +export interface GpuMetricStatRow { + gpuIndex: number; + metric: string; + count: number; + min: number; + max: number; + mean: number; + median: number; + p95: number; + p99: number; + stddev: number; +} + +export interface GpuMetricSeries { + id: number; + artifactName: string; + configKey: string; + fileName: string; + vendor: string; + sampleIntervalS: number | null; + sampleCount: number; + gpuCount: number; + startedAt: string; + endedAt: string; + sidecars: Record; + benchmarkResultIds: number[]; + /** Existing benchmark provenance can recover windows from older ingests. */ + powerAudits?: Record[]; + stats: GpuMetricStatRow[]; + data: GpuMetricSampleRow[]; +} + +export interface GpuMetricsRunPayload { + workflowRun: { + id: number; + githubRunId: number; + runAttempt: number; + name: string; + date: string; + htmlUrl: string | null; + headBranch: string | null; + headSha: string | null; + conclusion: string | null; + status: string | null; + createdAt: string | null; + }; + series: GpuMetricSeries[]; + /** Complete explorer labels when only one view artifact's samples were requested. */ + artifactNames?: string[]; +} + +export interface GpuMetricsRunSelection { + prefix?: string | null; + sourceResults?: readonly string[] | null; + /** Undefined reads all matching artifacts; null selects the first explorer artifact. */ + artifact?: string | null; +} + +// One sample row serializes to roughly 0.5 KB over the Neon HTTP driver (17 numeric +// columns plus keys and an ISO timestamp), so 50k rows stay near 25 MB, well below the +// 64 MiB response cap, while a 1.25M-sample run needs ~25 sequential pages, not 125. +export const SAMPLE_PAGE_SIZE = 50_000; + +interface RawSeriesRow { + id: number | string; + workflow_run_id: number | string; + artifact_name: string; + config_key: string; + file_name: string; + vendor: string; + sample_interval_s: number | null; + sample_count: number; + gpu_count: number; + started_at: string | Date; + ended_at: string | Date; + sidecars: Record | string; + benchmark_result_ids: (number | string)[] | null; + power_audits: Record[] | null; + csv_sha256: string; + ingested_at: string | Date; + stats_version: number; +} + +/** The columns `upsertGpuMetricSeries` rewrites whenever it replaces a series. */ +interface RawSeriesVersionRow { + id: number | string; + csv_sha256: string; + sample_count: number; + ingested_at: string | Date; + stats_version: number; +} + +interface RawStatRow { + series_id: number | string; + gpu_index: number; + metric: string; + sample_count: number; + min_value: number; + max_value: number; + mean_value: number; + median_value: number; + p95_value: number; + p99_value: number; + stddev_value: number; +} + +interface RawSampleRow extends StoredGpuMetricSample { + series_id: number | string; +} + +const isoString = (value: string | Date): string => + value instanceof Date ? value.toISOString() : new Date(value).toISOString(); + +const optional = (value: number | null): number | undefined => (value === null ? undefined : value); + +/** + * Thrown by the point/run readers when a series was replaced while its parts + * were being read; the caller restarts the whole read (links may have changed). + */ +export class TelemetrySnapshotChangedError extends Error { + constructor(seriesIds: readonly number[]) { + super(`gpu_metric_series ${seriesIds.join(', ')} changed while being read`); + this.name = 'TelemetrySnapshotChangedError'; + } +} + +const MAX_SNAPSHOT_ATTEMPTS = 3; + +const seriesVersionKey = (row: RawSeriesVersionRow): string => + `${Number(row.id)}:${row.csv_sha256}:${Number(row.sample_count)}:${isoString(row.ingested_at)}:${row.stats_version}`; + +async function withConsistentSnapshot(read: () => Promise): Promise { + for (let attempt = 1; ; attempt += 1) { + try { + return await read(); + } catch (error) { + if (!(error instanceof TelemetrySnapshotChangedError) || attempt >= MAX_SNAPSHOT_ATTEMPTS) { + throw error; + } + } + } +} + +function toSampleRow(raw: RawSampleRow): GpuMetricSampleRow { + return { + timestamp: isoString(raw.sampled_at), + index: Number(raw.gpu_index), + power: raw.power_w ?? 0, + temperature: optional(raw.temperature_c), + smClock: optional(raw.sm_clock_mhz), + memClock: optional(raw.mem_clock_mhz), + gpuUtil: optional(raw.gpu_util_pct), + memUtil: optional(raw.mem_util_pct), + edgeTemp: optional(raw.edge_temp_c), + memTemp: optional(raw.mem_temp_c), + gfxVoltage: optional(raw.gfx_voltage_mv), + socVoltage: optional(raw.soc_voltage_mv), + memVoltage: optional(raw.mem_voltage_mv), + fclk: optional(raw.fclk_mhz), + socClk: optional(raw.socclk_mhz), + mmActivity: optional(raw.mm_activity_pct), + }; +} + +function toStatRow(raw: RawStatRow): GpuMetricStatRow { + return { + gpuIndex: Number(raw.gpu_index), + metric: raw.metric, + count: Number(raw.sample_count), + min: raw.min_value, + max: raw.max_value, + mean: raw.mean_value, + median: raw.median_value, + p95: raw.p95_value, + p99: raw.p99_value, + stddev: raw.stddev_value, + }; +} + +async function loadSeriesDetails( + sql: DbClient, + seriesRows: readonly RawSeriesRow[], +): Promise { + if (seriesRows.length === 0) return []; + const ids = seriesRows.map((row) => Number(row.id)); + + const statRows = (await sql` + select series_id, gpu_index, metric, sample_count, min_value, max_value, mean_value, + median_value, p95_value, p99_value, stddev_value + from gpu_metric_gpu_stats + where series_id = any(${ids}::bigint[]) + order by series_id, gpu_index, metric + `) as unknown as RawStatRow[]; + + const sampleRows: RawSampleRow[] = []; + let cursor: RawSampleRow | undefined; + for (;;) { + const page = (await sql` + select series_id, gpu_index, + to_char(sampled_at at time zone 'UTC', 'YYYY-MM-DD"T"HH24:MI:SS.US"Z"') as sampled_at, + power_w, temperature_c, sm_clock_mhz, mem_clock_mhz, gpu_util_pct, mem_util_pct, + edge_temp_c, mem_temp_c, gfx_voltage_mv, soc_voltage_mv, mem_voltage_mv, + fclk_mhz, socclk_mhz, mm_activity_pct + from gpu_metric_samples + where series_id = any(${ids}::bigint[]) + and (${cursor?.series_id ?? null}::bigint is null or + (series_id, gpu_index, sampled_at) > + (${cursor?.series_id ?? null}::bigint, ${cursor?.gpu_index ?? null}::smallint, + ${cursor?.sampled_at ?? null}::timestamptz)) + order by series_id, gpu_index, gpu_metric_samples.sampled_at + limit ${SAMPLE_PAGE_SIZE} + `) as unknown as RawSampleRow[]; + sampleRows.push(...page); + if (page.length < SAMPLE_PAGE_SIZE) break; + // Keep PostgreSQL's microseconds: a JS Date cursor would repeat or omit boundary rows. + cursor = page.at(-1)!; + } + // The primary-key pages group by GPU; retain the public time-then-GPU ordering. + sampleRows.sort( + (a, b) => + Number(a.series_id) - Number(b.series_id) || + String(a.sampled_at).localeCompare(String(b.sampled_at)) || + Number(a.gpu_index) - Number(b.gpu_index), + ); + + // The series, stats and sample pages use separate autocommit queries (the DbClient + // has no transaction), so a re-ingest can commit between them. The writer + // replaces samples, stats and the series row in one transaction and stamps + // `ingested_at`; digest-only upgrades change `stats_version`. An unchanged + // key after the sample read proves both belong to the same series version. + const versionRows = (await sql` + select id, csv_sha256, sample_count, ingested_at, + coalesce((to_jsonb(s)->>'stats_version')::integer, 0) as stats_version + from gpu_metric_series s + where id = any(${ids}::bigint[]) + `) as unknown as RawSeriesVersionRow[]; + const versionKeys = new Set(versionRows.map(seriesVersionKey)); + if ( + versionRows.length !== seriesRows.length || + !seriesRows.every((row) => versionKeys.has(seriesVersionKey(row))) + ) { + throw new TelemetrySnapshotChangedError(ids); + } + + const statsBySeries = new Map(); + for (const raw of statRows) { + const key = Number(raw.series_id); + const bucket = statsBySeries.get(key); + if (bucket) bucket.push(toStatRow(raw)); + else statsBySeries.set(key, [toStatRow(raw)]); + } + const staleIds = new Set( + seriesRows + .filter((row) => row.stats_version !== GPU_STATS_VERSION) + .map((row) => Number(row.id)), + ); + const staleSamples = new Map(); + for (const sample of sampleRows) { + const id = Number(sample.series_id); + if (!staleIds.has(id)) continue; + const rows = staleSamples.get(id) ?? []; + rows.push(sample); + staleSamples.set(id, rows); + } + for (const row of seriesRows) { + const id = Number(row.id); + if (!staleIds.has(id)) continue; + const samples = staleSamples.get(id) ?? []; + if (samples.length !== Number(row.sample_count)) { + // Keep samples available to the existing source-gap recovery path, but + // never present a partial-population digest as full-record statistics. + statsBySeries.set(id, []); + continue; + } + statsBySeries.set( + id, + computeStoredGpuMetricStats(samples).map((stat) => ({ + ...stat, + metric: statMetricColumn(stat.metric), + })), + ); + } + const samplesBySeries = new Map(); + for (const raw of sampleRows) { + const key = Number(raw.series_id); + const bucket = samplesBySeries.get(key); + if (bucket) bucket.push(toSampleRow(raw)); + else samplesBySeries.set(key, [toSampleRow(raw)]); + } + + return seriesRows.map((row) => { + const id = Number(row.id); + return { + id, + artifactName: row.artifact_name, + configKey: row.config_key, + fileName: row.file_name, + vendor: row.vendor, + sampleIntervalS: row.sample_interval_s, + sampleCount: Number(row.sample_count), + gpuCount: Number(row.gpu_count), + startedAt: isoString(row.started_at), + endedAt: isoString(row.ended_at), + sidecars: + typeof row.sidecars === 'string' + ? (JSON.parse(row.sidecars) as Record) + : row.sidecars, + benchmarkResultIds: (row.benchmark_result_ids ?? []).map(Number), + powerAudits: row.power_audits ?? [], + stats: statsBySeries.get(id) ?? [], + data: samplesBySeries.get(id) ?? [], + }; + }); +} + +/** + * Stored telemetry for one GitHub Actions run (latest attempt), optionally scoped + * before reading samples. View selection also returns the full artifact-name inventory. + * Returns null when the run is unknown or has no stored series, so callers can + * fall back to the live GitHub artifacts for in-flight runs. + */ +export function getGpuMetricsForRun( + sql: DbClient, + githubRunId: number, + selection: GpuMetricsRunSelection = {}, +): Promise { + return withConsistentSnapshot(() => readGpuMetricsForRun(sql, githubRunId, selection)); +} + +async function readGpuMetricsForRun( + sql: DbClient, + githubRunId: number, + selection: GpuMetricsRunSelection, +): Promise { + const runRows = (await sql` + select id, github_run_id, run_attempt, name, date, html_url, head_branch, head_sha, + conclusion, status, created_at + from workflow_runs + where github_run_id = ${githubRunId} + order by run_attempt desc + limit 1 + `) as unknown as { + id: number | string; + github_run_id: number | string; + run_attempt: number; + name: string; + date: string | Date; + html_url: string | null; + head_branch: string | null; + head_sha: string | null; + conclusion: string | null; + status: string | null; + created_at: string | Date | null; + }[]; + const run = runRows[0]; + if (!run) return null; + + let artifactNames: string[] | undefined; + let selectedIds: number[] | null = null; + if (selection.artifact !== undefined) { + const inventory = (await sql` + select id, + case when count(*) over (partition by artifact_name) > 1 + then artifact_name || '/' || file_name else artifact_name end as name + from gpu_metric_series + where workflow_run_id = ${Number(run.id)} + order by artifact_name, file_name + `) as { id: number | string; name: string }[]; + if (inventory.length === 0) return null; + artifactNames = inventory.map((entry) => entry.name); + const selected = selection.artifact ?? artifactNames[0]; + selectedIds = inventory + .filter((entry) => entry.name === selected) + .map((entry) => Number(entry.id)); + } + const prefix = selection.prefix ?? null; + const sources = selection.sourceResults ?? null; + + const seriesRows = (await sql` + select s.id, s.workflow_run_id, s.artifact_name, s.config_key, s.file_name, s.vendor, + s.sample_interval_s, s.sample_count, s.gpu_count, s.started_at, s.ended_at, s.sidecars, + s.csv_sha256, s.ingested_at, + coalesce((to_jsonb(s)->>'stats_version')::integer, 0) as stats_version, + ( + select array_agg(l.benchmark_result_id order by l.benchmark_result_id) + from benchmark_result_gpu_metrics l where l.series_id = s.id + ) as benchmark_result_ids, + ( + select jsonb_agg(br.power_audit order by br.id) + from benchmark_result_gpu_metrics l + join benchmark_results br on br.id = l.benchmark_result_id + where l.series_id = s.id and br.power_audit is not null + ) as power_audits + from gpu_metric_series s + where s.workflow_run_id = ${Number(run.id)} + and (${selectedIds}::bigint[] is null or s.id = any(${selectedIds}::bigint[])) + and (${prefix}::text is null or starts_with(s.artifact_name, 'gpu_metrics_' || ${prefix}) + or (starts_with(s.artifact_name, 'power_audit_') and + (starts_with(s.artifact_name, 'power_audit_' || ${prefix}) + or starts_with('power_audit_' || ${prefix}, s.artifact_name)))) + and (${sources}::text[] is null or exists ( + select 1 from unnest(${sources}::text[]) as requested(result) + where s.artifact_name = 'gpu_metrics_' || requested.result + or (starts_with(s.artifact_name, 'power_audit_') and + (starts_with(s.artifact_name, 'power_audit_' || requested.result) + or starts_with('power_audit_' || requested.result, s.artifact_name))) + )) + order by s.artifact_name, s.file_name + `) as unknown as RawSeriesRow[]; + if (seriesRows.length === 0 && artifactNames === undefined) return null; + + return { + workflowRun: { + id: Number(run.id), + githubRunId: Number(run.github_run_id), + runAttempt: Number(run.run_attempt), + name: run.name, + date: isoString(run.date).slice(0, 10), + htmlUrl: run.html_url, + headBranch: run.head_branch, + headSha: run.head_sha, + conclusion: run.conclusion, + status: run.status, + createdAt: run.created_at ? isoString(run.created_at) : null, + }, + series: await loadSeriesDetails(sql, seriesRows), + ...(artifactNames === undefined ? {} : { artifactNames }), + }; +} + +export interface GpuMetricsPointPayload { + benchmarkResultId: number; + series: GpuMetricSeries[]; +} + +/** Series linked to one benchmark point, with samples. Null when none is linked. */ +export function getGpuMetricsForPoint( + sql: DbClient, + benchmarkResultId: number, +): Promise { + return withConsistentSnapshot(() => readGpuMetricsForPoint(sql, benchmarkResultId)); +} + +async function readGpuMetricsForPoint( + sql: DbClient, + benchmarkResultId: number, +): Promise { + const seriesRows = (await sql` + select s.id, s.workflow_run_id, s.artifact_name, s.config_key, s.file_name, s.vendor, + s.sample_interval_s, s.sample_count, s.gpu_count, s.started_at, s.ended_at, s.sidecars, + s.csv_sha256, s.ingested_at, + coalesce((to_jsonb(s)->>'stats_version')::integer, 0) as stats_version, + ( + select array_agg(l.benchmark_result_id order by l.benchmark_result_id) + from benchmark_result_gpu_metrics l where l.series_id = s.id + ) as benchmark_result_ids, + ( + select jsonb_agg(br.power_audit order by br.id) + from benchmark_result_gpu_metrics l + join benchmark_results br on br.id = l.benchmark_result_id + where l.series_id = s.id and br.power_audit is not null + ) as power_audits + from benchmark_result_gpu_metrics link + join gpu_metric_series s on s.id = link.series_id + where link.benchmark_result_id = ${benchmarkResultId} + order by s.artifact_name, s.file_name + `) as unknown as RawSeriesRow[]; + if (seriesRows.length === 0) return null; + return { + benchmarkResultId, + series: await loadSeriesDetails(sql, seriesRows), + }; +} diff --git a/packages/db/src/verify-db.ts b/packages/db/src/verify-db.ts index 1f4d51edd..3ab8f805c 100644 --- a/packages/db/src/verify-db.ts +++ b/packages/db/src/verify-db.ts @@ -50,6 +50,9 @@ async function verify(): Promise { TABLE_NAMES.evalResults, TABLE_NAMES.evalSamples, TABLE_NAMES.changelogEntries, + TABLE_NAMES.gpuMetricSeries, + TABLE_NAMES.gpuMetricSamples, + TABLE_NAMES.benchmarkResultGpuMetrics, ]; for (const t of tables) { const [{ n }] = await sql`select count(*)::int as n from ${sql(t)}`; diff --git a/packages/db/src/verify-power-publication.ts b/packages/db/src/verify-power-publication.ts index 84bf86723..41038b292 100644 --- a/packages/db/src/verify-power-publication.ts +++ b/packages/db/src/verify-power-publication.ts @@ -1,7 +1,9 @@ import fs from 'node:fs'; import { DB_MODEL_TO_DISPLAY } from '@semianalysisai/inferencex-constants'; import { createAdminSql } from './etl/db-utils'; +import { verifyTelemetryApi } from './etl/telemetry-receipt'; import { + fatalPublicationErrors, verifyPowerPublication, type PowerPublicationManifest, type PublishedPowerRow, @@ -33,12 +35,12 @@ try { join workflow_runs wr on wr.id = br.workflow_run_id where wr.github_run_id = ${manifest.runId} and wr.run_attempt = ${manifest.runAttempt} and (br.benchmark_type = 'agentic_traces' or - (br.benchmark_type = 'single_turn' and br.isl = 8192 and br.osl = 1024)) + (br.benchmark_type = 'single_turn' and br.isl in (1024, 8192) and br.osl = 1024)) `; - const errors = [ - ...(manifest.ingestErrors ?? []), - ...verifyPowerPublication(manifest.points, rows as unknown as PublishedPowerRow[], 'database'), - ]; + const errors = fatalPublicationErrors( + manifest, + verifyPowerPublication(manifest.points, rows as unknown as PublishedPowerRow[], 'database'), + ); const publicRows: PublishedPowerRow[] = []; const models = [ ...new Set(manifest.points.map((point) => DB_MODEL_TO_DISPLAY[String(point.identity.model)])), @@ -70,6 +72,19 @@ try { publicRows.push(...body); } errors.push(...verifyPowerPublication(manifest.points, publicRows, 'public API')); + if (manifest.telemetry) { + if ( + manifest.telemetry.runId !== manifest.runId || + manifest.telemetry.runAttempt !== manifest.runAttempt + ) + throw new Error('Telemetry receipt belongs to a different run/attempt'); + manifest.telemetry = await verifyTelemetryApi(manifest.telemetry, origin, { + headers: process.env.CACHE_PROTECTION_BYPASS_SECRET + ? { 'x-vercel-protection-bypass': process.env.CACHE_PROTECTION_BYPASS_SECRET } + : undefined, + }); + fs.writeFileSync(manifestPath, `${JSON.stringify(manifest, null, 2)}\n`); + } const counts = { strict: 0, invalid: 0, other: 0 }; for (const point of manifest.points) { if (point.metrics.power_valid === 1 && point.metrics.power_metric_schema_version === 2) @@ -86,6 +101,10 @@ try { status: errors.length > 0 ? 'failed' : manifest.points.length > 0 ? 'matched' : 'no_power_points', errors, + // Kept out of `errors` on purpose: a telemetry digest failure costs one + // point's PowerX tab, not its benchmark data, so it must not fail the ingest. + telemetryWarnings: manifest.telemetryWarnings ?? [], + ...(manifest.telemetry ? { telemetry: manifest.telemetry } : {}), }; fs.writeFileSync(`${manifestPath}.verification.json`, `${JSON.stringify(receipt, null, 2)}\n`); console.log(JSON.stringify(receipt, null, 2)); diff --git a/packages/skills/skills/inferencex-api/integrity.json b/packages/skills/skills/inferencex-api/integrity.json index b9389a7d7..9127fea35 100644 --- a/packages/skills/skills/inferencex-api/integrity.json +++ b/packages/skills/skills/inferencex-api/integrity.json @@ -8,10 +8,10 @@ "references/cli-contract.md": "fa44faa38d889b4fbdee5ba42752b47758ee3706fab150513bd4e6db87a86cb0", "references/cli.md": "96b228f34cb3600f4548f4dc84df8506531747286763c56ca834c77bee05e1eb", "references/collectivex.md": "eb794f9c28d4a27bec4db80c42c4685ff3b1d204fd9b12a6788511fb258a789f", - "references/dashboard-views.md": "57d78ece6595c3b8a28bf09722a633b3fe8e9e1322046534856d1f34e30ee805", + "references/dashboard-views.md": "339ebf9e7b33e4d586c852452a42a8ccf020e7edc0427f434d51cd092e69486e", "references/offline-exports.md": "95aa565dcd4c9e592159baa560a9a58217bc61e395de1c6daf0576795c7f4ec9", "references/pareto.md": "1b4d2d163f982e3f2d97addce310789eae9c1e5badd82dd7501708c9e0385027", - "references/powerx.md": "cf8adcc5dfe395c5fb659908819980a872475f862f22b723815af43d363cedad", + "references/powerx.md": "ae01164107aec35247f729e425379b350d2e7ecdd8d4668fc68dbd879f3d965d", "references/provenance.md": "da4e9d078b31f234727706593f77ca9d5fdac611921783aaa09b5741de58b69a", "references/public-api-examples.md": "dbefcbbaa17c1aedf922e22a209b22511020e79bfe5c01e0b8b69dc4fc263e35", "references/releases.md": "b858c44360dc9fc394b11c7318245b167bc6d0f0f95e46c68f6990eaf90496d8", diff --git a/packages/skills/skills/inferencex-api/references/dashboard-views.md b/packages/skills/skills/inferencex-api/references/dashboard-views.md index 6f97cf3c5..81f123044 100644 --- a/packages/skills/skills/inferencex-api/references/dashboard-views.md +++ b/packages/skills/skills/inferencex-api/references/dashboard-views.md @@ -59,7 +59,7 @@ which controls belong together. | `collectivex` | Ordered run selection and suite; EP size/phase/modes/precision/operation/percentile/axis/SKU/backend/series; KV page size/x/y/pull/push/overlap ISL/series; swap direction/layout/metric/percentile/series. | | `submissions` | Search, table sort/direction/offset/limit, weekly/cumulative chart, on-change cutoff and NVIDIA/AMD/total lines. Table search does not filter the independent submission-volume chart. | | `current-inferencex-image` | Model, sequence, hardware, precision, speculation, node type, framework families and `asOf` for image age/release status. | -| `gpu-metrics` | Required run, artifact, GPU indices, metric or correlation axes, statistics sorting, chart mode and interactive downsampling preference. Raw rows and statistics remain unsampled; live response is no-store. | +| `gpu-metrics` | Required run, file/host artifact, GPU indices, metric or correlation axes, statistics sorting, chart mode and interactive downsampling preference. DB-first telemetry; full-record statistics use stored digests. Responses are no-store. | | `video` | CI run/artifact discovery; only already-published artifact reads; source, serving cell, media/fidelity slot, power phase and GPU denominator; same-workload comparisons, axes, deployment costs and selected tradeoff point. No-store. | ## Interpretation and maintenance @@ -94,8 +94,15 @@ Custom chip costs are USD/chip-hour, token prices USD/million tokens, interactivity tok/s/user, and video costs USD/deployment-hour. Preserve the response's resolved TCO, utilization, license share, topology and power basis. Modeled and provisioned power are distinct. Missing measured evidence is not zero. -GPU chart projections preserve the existing renderer's missing-metric zero -fallback; use unsampled `rows` to distinguish absent optional sensor values. +GPU chart projections omit missing metric readings; measured zero remains zero. +The selected file/host series' full-record statistics include startup and warmup +and cover all chips regardless of visibility or chart downsampling. Stats rows omit +the storage-only `metric` column; `params.metric` identifies the selected metric. Current-version stored digests +are authoritative, including empty or absent metric digests. Outdated or +unversioned digests are recomputed read-only from retained DB samples. Keep these sample-weighted statistics separate from +serving-window power, J/token and selected-time-window calculations. The view reads +stored telemetry first, falls back to artifacts for missing storage, and preserves +upstream 503 failures. Treat an error as unavailable evidence, not an empty dataset. Zoom, axis scale, theme, labels, report expansion, media playback and download buttons are presentation state, not new datasets. AI-chart provider keys and diff --git a/packages/skills/skills/inferencex-api/references/powerx.md b/packages/skills/skills/inferencex-api/references/powerx.md index ca69168ae..c2b3bf975 100644 --- a/packages/skills/skills/inferencex-api/references/powerx.md +++ b/packages/skills/skills/inferencex-api/references/powerx.md @@ -57,6 +57,17 @@ Existing output directories are refused. Use `inferencex verify ` to reconstruct the result from the saved responses. A later API request is separate evidence and may return different data. +## NVL72 CPU measurements + +Raw benchmark responses may include `cpu_power_valid`, `avg_cpu_socket_power_w`, +`avg_total_cpu_power_w`, `total_cpu_energy_j`, and optional module power/energy. +CPU measurements require `cpu_power_valid === 1`, independently of GPU validity. +`powerValid=strictV2` filters GPU measurements only. Read `power_audit.cpu` for +sensor type, collector, socket coverage, and validation reasons. Grace socket +readings combine CPU and LPDDR5X power; a module reading already includes GPU +power. Missing measurements stay unavailable. These fields do not add CPU +samples to the GPU timeline or change the CLI export format. + ## Selection and coverage The request uses `/api/v1/benchmarks` with `model`, optional `date`, and