diff --git a/.github/linters/.markdown-lint.yml b/.github/linters/.markdown-lint.yml new file mode 100644 index 000000000..08a64e071 --- /dev/null +++ b/.github/linters/.markdown-lint.yml @@ -0,0 +1,14 @@ +# Super-Linter's default markdownlint rules (v8.7.0 TEMPLATES/.markdown-lint.yml) +# with MD013 line length off. Super-Linter lints each changed file in full, and +# CHANGELOG.md keeps each entry on one line while docs tables have rows past +# 400 characters, so any edit to those files failed on lines nobody touched. +MD004: false +MD007: + indent: 2 +MD013: false +MD026: + punctuation: ".,;:!。,;:" +MD029: false +MD033: false +MD036: false +blank_lines: false diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index fb1b98a4f..97fde79a0 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -16,18 +16,18 @@ on: permissions: contents: read +# A new push to a pull request cancels that pull request's older run. Every +# other run gets its own group: a shared group would still cancel a queued run +# on main whenever another push arrived. concurrency: - group: ci-${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} + group: ci-${{ github.workflow }}-${{ github.event.pull_request.number || github.run_id }} cancel-in-progress: true jobs: draft-and-build-gates: - name: "${{ github.event_name == 'pull_request' && 'PR build and Node.js 24 runtime smoke' || 'Build gates' }}" + name: Build gates runs-on: ubuntu-latest - # The PR path also runs the serial runtime and CLI contract smoke suites. - # Under runner contention they completed successfully just after the former - # 15-minute ceiling, so keep a bounded two-times completion budget. - timeout-minutes: 30 + timeout-minutes: 15 steps: - name: Check out repository @@ -67,6 +67,13 @@ jobs: - name: Find dead code and dependencies run: pnpm -w knip + # Informational until the pending dead-code removals land; then these + # categories move into the blocking `pnpm -w knip` step above. Findings + # exit 0, so they do not add a failure annotation to every run. + - name: Report unused exports + continue-on-error: true + run: pnpm -w knip --include exports,types,duplicates --no-exit-code + - name: Build run: pnpm -w build @@ -84,14 +91,6 @@ jobs: - name: Enforce production dependency advisory policy run: pnpm -w security:dependency-advisories - - name: Run PR runtime smoke tests - if: github.event_name == 'pull_request' - run: pnpm --filter @ultrafuzz/runtime test:pr-smoke:prebuilt - - - name: Run PR CLI status contract smoke tests - if: github.event_name == 'pull_request' - run: pnpm --filter @ultrafuzz/cli test:pr-smoke:prebuilt - external-static-analysis: name: External static analysis runs-on: ubuntu-latest @@ -125,10 +124,10 @@ jobs: VALIDATE_SHELL_SHFMT: true VALIDATE_YAML: true - # The lane table lives in scripts/ci/release-validation-lanes.mjs so the - # pull-request policy is a tested artifact instead of an invisible `if:`. - # While this job was gated off pull requests, every runtime test reported - # `skipping` on a PR and resume-path regressions merged with all checks green. + # The lane table lives in scripts/ci/release-validation-lanes.mjs, whose test + # checks that it runs every gate scripts/validate-release.mjs defines. Every + # lane runs on pull requests too: while lanes were gated off pull requests, + # regressions in the skipped suites merged with all checks green. release-validation-lanes: name: Select release validation lanes runs-on: ubuntu-latest @@ -144,17 +143,14 @@ jobs: - name: Select release validation lanes id: select - env: - EVENT_NAME: ${{ github.event_name }} run: | - echo "lanes=$(node scripts/ci/release-validation-lanes.mjs --event "$EVENT_NAME")" >> "$GITHUB_OUTPUT" + echo "lanes=$(node scripts/ci/release-validation-lanes.mjs)" >> "$GITHUB_OUTPUT" + # The lanes start without waiting for the build gates; release-gates still + # requires every job. release-validation: name: Full release validation (${{ matrix.description }}) - needs: - - draft-and-build-gates - - external-static-analysis - - release-validation-lanes + needs: release-validation-lanes runs-on: ubuntu-latest timeout-minutes: ${{ matrix.timeout_minutes }} strategy: @@ -172,6 +168,14 @@ jobs: sudo rm -rf /usr/local/lib/android sudo rm -rf /usr/share/dotnet + # Every test that launches a run copies a sealed execution snapshot of + # thousands of files and fsyncs each one in the test process, where + # eatmydata makes fsync a no-op. The Smithers engine's environment is + # built from an allowlist without LD_PRELOAD, so engine processes still + # sync. An ephemeral runner has nothing to protect across a crash. + - name: Install eatmydata + run: command -v eatmydata || { sudo apt-get update && sudo apt-get install -y eatmydata; } + - name: Check out repository uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: @@ -214,7 +218,7 @@ jobs: - name: Validate release lane run: >- - pnpm -w validate:release -- + eatmydata pnpm -w validate:release -- --gates "${{ matrix.gates }}" --report ".ultrafuzz/release-validation/${{ matrix.lane }}.json" diff --git a/.ultrafuzz/evals/bug-finding.yml b/.ultrafuzz/evals/bug-finding.yml index 96fdf8d4c..7302e11ab 100644 --- a/.ultrafuzz/evals/bug-finding.yml +++ b/.ultrafuzz/evals/bug-finding.yml @@ -45,7 +45,7 @@ reporting: # policy only — provider selection/credentials live in ultrafuzz.to node_telemetry: true heartbeat_interval_seconds: 60 experiment_prefix: bug-finding - artifacts: - mode: manifest-only # forced default when target.sensitivity == private; opt in to "upload" + artifacts: # still validated, but no eval behaviour depends on it + mode: manifest-only include: ["report.md", "report.json"] max_file_bytes: 5000000 diff --git a/.ultrafuzz/prompts/_templates/output-contract/coverage-evidence-markdown.mdx b/.ultrafuzz/prompts/_templates/output-contract/coverage-evidence-markdown.mdx index f963a7f16..050533c23 100644 --- a/.ultrafuzz/prompts/_templates/output-contract/coverage-evidence-markdown.mdx +++ b/.ultrafuzz/prompts/_templates/output-contract/coverage-evidence-markdown.mdx @@ -37,5 +37,6 @@ Blockers: Repeat blocker and evidence rows in artifact order. Runtime publication compares this section with the typed handoff and rejects missing, duplicated, reordered, -or bare coverage scores. Raw `covg-eval` output is for iteration only and -defines neither published declaration-completeness view. +or contradicting scoped scores, and it warns about coverage scores that name no +exact scope. Raw `covg-eval` output is for iteration only and defines neither +published declaration-completeness view. diff --git a/.ultrafuzz/prompts/review/final-report.md b/.ultrafuzz/prompts/review/final-report.md index 70008c7ae..ab9a4965c 100644 --- a/.ultrafuzz/prompts/review/final-report.md +++ b/.ultrafuzz/prompts/review/final-report.md @@ -385,25 +385,8 @@ Ultrafuzz is an automated smart-contract fuzzing campaign assistant. Issues belo - Tokens used: `` - Estimated spend: `` - Audit profile: `` - -## Audit context - -- Threat model: [THREAT_MODEL.md](); [threat-model.json]() -- Goal plan: [goal-plan.json]() ``` -Render `## Audit context` with exactly this heading, bullet order, and link -text, immediately after `## Run summary`. Use repository-relative or -report-relative paths to the run's own `threat-model` and `goal-plan` artifacts; -never absolute paths or external URLs. Omit an individual link whose artifact -the run did not produce, omit the `Goal plan` bullet when there is no goal plan, -and omit the whole section when the run produced none of them. Do not invent a -different heading, ordering, or link text: `ultrafuzz report` regenerates this -exact section deterministically from the run's own artifacts and overwrites -anything else. -Keep detailed threat content in those dedicated artifacts; do not duplicate it -in `report.md`. - Each production issue entry must use exactly this Markdown section order. The following example is structural only; replace the title, actor names, actions, outcomes, explanations, code, variants, and strategy IDs with issue-specific @@ -825,8 +808,6 @@ Before finishing, verify that: - `report.md` contains `## Property provenance`, including every property-derived finding and no invented property IDs for non-property findings. -- `report.md` renders the fixed `## Audit context` section for every artifact - the run produced, without copying their detailed analysis. - `report.md` contains `## Property implementation coverage` rendered from the exact runtime-authoritative coverage object. - `report.md` contains `## Goal search coverage` with counts recomputed from the diff --git a/.ultrafuzz/prompts/setup/discover-base-test.md b/.ultrafuzz/prompts/setup/discover-base-test.md index 64ccf3ecc..1d9563348 100644 --- a/.ultrafuzz/prompts/setup/discover-base-test.md +++ b/.ultrafuzz/prompts/setup/discover-base-test.md @@ -39,8 +39,10 @@ If the setup handoffs identify Vyper-only or mixed Solidity/Vyper production contracts, make the reusable Foundry fixture Vyper-aware while keeping the tests Solidity-based. Define Solidity interfaces for the Vyper contracts' ABI-visible public/external functions and events, or reuse ABI-derived interfaces generated -by the target repository. Do not require Foundry to compile `.vy` files as -Solidity sources. +by the target repository. Declare any new interface in the test tree: the +workspace handoff rejects every change under the production source roots (by +default `src/` and `contracts/`). Do not require Foundry to compile `.vy` files +as Solidity sources. For Vyper deployment helpers, prefer one reusable path that compiles creation bytecode with the target project's pinned compiler/tooling from the project diff --git a/.ultrafuzz/prompts/setup/prepare-foundry-harness.md b/.ultrafuzz/prompts/setup/prepare-foundry-harness.md index 0cc1e14ac..226ab1644 100644 --- a/.ultrafuzz/prompts/setup/prepare-foundry-harness.md +++ b/.ultrafuzz/prompts/setup/prepare-foundry-harness.md @@ -22,7 +22,9 @@ production contracts, keep Foundry as the test harness and do not ask Foundry to compile `.vy` files as Solidity sources. Configure the harness so generated `.t.sol` tests interact with Vyper contracts through Solidity interfaces that match the contracts' public/external ABI, or through ABI-derived Solidity -interfaces when the target repository already generates them. +interfaces when the target repository already generates them. Declare any new +interface in the test tree: the workspace handoff rejects every change under the +production source roots (by default `src/` and `contracts/`). Create only the minimal harness layout needed by later fuzzing agents. diff --git a/.ultrafuzz/prompts/strategies/invariants/implement-properties.md b/.ultrafuzz/prompts/strategies/invariants/implement-properties.md index 4e7aa6759..4d2180b4c 100644 --- a/.ultrafuzz/prompts/strategies/invariants/implement-properties.md +++ b/.ultrafuzz/prompts/strategies/invariants/implement-properties.md @@ -124,8 +124,10 @@ an unselected expected check is not implemented or fulfilled. guard, or other precondition that prevents the backend from observing the violating post-state. Preconditions may admit valid actions; they may not assume the property under test. - - Do not edit production contracts except interfaces that are genuinely - required by the test harness. + - Do not edit production contracts, not even to add an interface: the + workspace handoff rejects every change under the production source roots + (by default `src/` and `contracts/`). Declare any interface the harness + needs in the test tree instead. - Keep generated or changed invariant files in the test tree and include every changed `*.t.sol` test/reproducer in `generated-tests.json`. - Preserve Recon constructor deployment if property work changes `Setup`, diff --git a/.ultrafuzz/prompts/strategies/invariants/setup.md b/.ultrafuzz/prompts/strategies/invariants/setup.md index 1ec988aee..eb0b641a5 100644 --- a/.ultrafuzz/prompts/strategies/invariants/setup.md +++ b/.ultrafuzz/prompts/strategies/invariants/setup.md @@ -71,7 +71,7 @@ project's pinned revision. Use these Recon/Chimera rules while making decisions: - Read `AGENTS.md` and obey all repository-specific rules before editing. -- Do not edit production `src/` or `contracts/` except for interfaces if they are genuinely required by the harness. +- Do not edit production `src/` or `contracts/`, not even to add an interface: the workspace handoff rejects every change under the production source roots. Declare any interface the harness needs in the test tree instead. - [Chimera](https://github.com/Recon-Fuzz/create-chimera-app) is the write-once, run-everywhere scaffold for Foundry, Echidna, Medusa, Halmos, and Kontrol style runs. - The create-chimera-app layout under the repository's test root is: `/recon/Setup.sol`, `BeforeAfter.sol`, `Properties.sol`, diff --git a/.ultrafuzz/topology.yml b/.ultrafuzz/topology.yml index 3e627b41d..c763c203a 100644 --- a/.ultrafuzz/topology.yml +++ b/.ultrafuzz/topology.yml @@ -41,6 +41,8 @@ groups: review: label: Review color: "#0f766e" + defaults: + timeout_seconds: 7200 nodes: - id: __start__ kind: meta @@ -198,7 +200,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/0kn0t.md contract: ultrafuzz/nonempty-markdown@1 @@ -211,7 +212,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/certora-thinking.md contract: ultrafuzz/nonempty-markdown@1 @@ -224,7 +224,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/certora-sanity.md contract: ultrafuzz/nonempty-markdown@1 @@ -237,7 +236,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/aviggiano.md contract: ultrafuzz/nonempty-markdown@1 @@ -250,7 +248,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/rounding.md contract: ultrafuzz/nonempty-markdown@1 @@ -263,7 +260,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/crytic.md contract: ultrafuzz/nonempty-markdown@1 @@ -276,7 +272,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/runtime-verification.md contract: ultrafuzz/nonempty-markdown@1 @@ -289,7 +284,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/a16z-erc4626.md contract: ultrafuzz/nonempty-markdown@1 @@ -302,7 +296,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/recon.md contract: ultrafuzz/nonempty-markdown@1 diff --git a/docs/SPECS.md b/docs/SPECS.md index 0fe0c4cea..215b3f13a 100644 --- a/docs/SPECS.md +++ b/docs/SPECS.md @@ -243,8 +243,15 @@ Every planned JSON output MUST resolve through the checked-in schema registry. The planned and expanded graph representations MUST persist the schema filename, fragment-free schema ID, schema SHA-256, package schema-bundle SHA-256, and validator build identity. Missing or partial bindings MUST fail planning or host -verification. Operators declare the versioned contract in topology; they MUST -NOT supply these trust identities manually in YAML. +verification. Host artifact gates and verified reads MUST validate against the +schema content a binding names, using the run's sealed schema snapshot when the +installed bundle differs. The recorded validator build and contract digest are +provenance: graph reads, host artifact gates, verified reads, task preparation, +dependency admission, and the validator preflight MUST NOT require them to equal +the reading build's own. Task preparation and the validator preflight MUST still +require the schema bundle they validate with to be the planned one. Operators +declare the versioned contract in topology; they MUST NOT supply these trust +identities manually in YAML. ## Prompts @@ -447,9 +454,10 @@ every non-builtin module whose lexical or physical resolution escapes the closure. Ordinary artifact and schema data reads remain outside this module boundary. Modal MUST provide the equivalent root-owned, read-only entrypoint. Both environments MUST run a real known-valid fixture and -verify the returned schema ID, schema digest, bundle digest, and validator build; -`command -v` alone is insufficient. A missing, tampered, or stale launcher or -closure is a setup failure for new model work. It MUST NOT turn historical +verify the returned schema ID, schema digest, and bundle digest; the returned +validator build is provenance and MUST NOT be compared. `command -v` alone is +insufficient. A missing, tampered, or stale launcher or closure is a setup +failure for new model work. It MUST NOT turn historical seals or schema identities into resume authorization. A current-controller continuation MAY select current validator packages while retaining historical source and artifacts as provenance. @@ -470,7 +478,6 @@ Before or at launch, each run MUST persist: - immutable rendered prompt snapshots under `prompt-snapshots/` - per-node artifacts under `artifacts/` - review artifacts under `review/` -- event query indexes under `events.index/` - workspace metadata under `workspaces/` Reporting is agentic and lives in final-report artifacts. diff --git a/docs/assets/eval-history/cost.svg b/docs/assets/eval-history/cost.svg deleted file mode 100644 index ea9d6eae2..000000000 --- a/docs/assets/eval-history/cost.svg +++ /dev/null @@ -1,779 +0,0 @@ - - -Cost (USD) -Cost (USD) by candidate commit and benchmark target; partial values use hollow dashed markers, legacy partial values without a number use a dashed ring and partial n/a label, and unavailable values use an n/a cross - -Cost (USD) -ultrafuzz-bench · smoke -Each line tracks one benchmark target across evenly spaced candidate-run columns. - - - -0 - -8.29 - -16.59 - -24.88 - -33.17 -ultrafuzz-bench smoke cohort-cf3da608 policy-b0f7d6d0 - -cohort-cf3da608 -ultrafuzz-bench smoke cohort-a98bf547 policy-33b8d296 - -cohort-a98bf547 -ultrafuzz-bench smoke cohort-7d061c80 policy-33b8d296 - -cohort-7d061c80 -ultrafuzz-bench smoke cohort-b05ba222 policy-33b8d296 - -cohort-b05ba222 -ultrafuzz-bench smoke cohort-c0e17843 policy-33b8d296 - -cohort-c0e17843 -ultrafuzz-bench smoke cohort-d49aa3e2 policy-33b8d296 - -cohort-d49aa3e2 -ultrafuzz-bench smoke cohort-5d6ac3fd policy-fbbf61e0 - -cohort-5d6ac3fd -ultrafuzz-bench smoke cohort-21f261d4 policy-dfea2bb8 - -cohort-21f261d4 -ultrafuzz-bench smoke cohort-f3e78e3d policy-a7379336 - -cohort-f3e78e3d -deepseek-v4-flash max stableswap-ng-vyper 6d1db74: 0.27 - - -deepseek-v4-flash max venus-isolated-pools-hardhat 6d1db74: 0.19 - - -deepseek-v4-flash max very-liquid-vaults-foundry 6d1db74: 0.22 - - -deepseek-v4-pro max stableswap-ng-vyper 09277b8: 0.43 - - -deepseek-v4-pro max venus-isolated-pools-hardhat 09277b8: 0.37 - - -deepseek-v4-pro max very-liquid-vaults-foundry 09277b8: 0.47 - - - -gpt-5.6-luna high stableswap-ng-vyper 953a397: 25.16 - - -gpt-5.6-luna high stableswap-ng-vyper 2051b64: 25.12 - - -gpt-5.6-luna high stableswap-ng-vyper 23ef040: 22.07 - - -gpt-5.6-luna high stableswap-ng-vyper 8483637: 25.53 - - -gpt-5.6-luna high stableswap-ng-vyper e03c176: 25.48 - - -gpt-5.6-luna high stableswap-ng-vyper da5d191: 20.7 - - -gpt-5.6-luna high stableswap-ng-vyper 24b90c4: 26.83 - - -gpt-5.6-luna high stableswap-ng-vyper 5cfa918: 12.13 - - -gpt-5.6-luna high stableswap-ng-vyper bdf2482: 29 - - -gpt-5.6-luna high stableswap-ng-vyper ed26197: 26.59 - - -gpt-5.6-luna high stableswap-ng-vyper 3ea3059: 21.12 - - -gpt-5.6-luna high stableswap-ng-vyper 63becee: 3.08 - - -gpt-5.6-luna high stableswap-ng-vyper bf9f729: 4.41 - - -gpt-5.6-luna high stableswap-ng-vyper 8aac39f: 4.24 - - -gpt-5.6-luna high stableswap-ng-vyper 80350a9: 4.18 - - -gpt-5.6-luna high stableswap-ng-vyper 4928cbe: 3.49 - - -gpt-5.6-luna high stableswap-ng-vyper e685be8: 3.98 - - -gpt-5.6-luna high stableswap-ng-vyper 10f5608: 6.33 - - -gpt-5.6-luna high stableswap-ng-vyper 04cfef8: 6.26 - - -gpt-5.6-luna high stableswap-ng-vyper fa8d039: 4.16 - - -gpt-5.6-luna high stableswap-ng-vyper 89e05a7: 4.32 - - -gpt-5.6-luna high stableswap-ng-vyper 0e073cf: 4.17 - - -gpt-5.6-luna high stableswap-ng-vyper 09deffa: 3.73 - - -gpt-5.6-luna high stableswap-ng-vyper a4b3c97: 3.43 - - -gpt-5.6-luna high stableswap-ng-vyper 650dc02: 3.95 - - -gpt-5.6-luna high stableswap-ng-vyper 4c59002: 3.17 - - -gpt-5.6-luna high stableswap-ng-vyper fd079e3: 4.64 - - -gpt-5.6-luna high stableswap-ng-vyper e750760: 4.98 - - -gpt-5.6-luna high stableswap-ng-vyper 5b7c0c0: 4.08 - - -gpt-5.6-luna high stableswap-ng-vyper b22f23c: 4.37 - - -gpt-5.6-luna high stableswap-ng-vyper 2d47e73: 3.35 - - -gpt-5.6-luna high stableswap-ng-vyper 5db8c6f: 3.7 - - -gpt-5.6-luna high stableswap-ng-vyper 650a87a: 2.93 - - -gpt-5.6-luna high stableswap-ng-vyper cb672c6: 2.82 - - -gpt-5.6-luna high stableswap-ng-vyper 6d1db74: 6.38 - - -gpt-5.6-luna high stableswap-ng-vyper dd874e4: 3.97 - - -gpt-5.6-luna high stableswap-ng-vyper 149db7d: 5.91 - - -gpt-5.6-luna high stableswap-ng-vyper ed59513: 4.71 - - -gpt-5.6-luna high stableswap-ng-vyper 12d716e: 4.39 - - -gpt-5.6-luna high stableswap-ng-vyper 1775fec: 3.73 - - -gpt-5.6-luna high stableswap-ng-vyper 3663089: 4.9 - - -gpt-5.6-luna high stableswap-ng-vyper 0f5117f: 4.93 - - -gpt-5.6-luna high stableswap-ng-vyper 87442ee: unavailable (pricing-unavailable) - - -n/a 87442ee -gpt-5.6-luna high stableswap-ng-vyper ac03a8d: 3.64 - - -gpt-5.6-luna high stableswap-ng-vyper 2a2efb2: unavailable (pricing-unavailable) - - -n/a 2a2efb2 -gpt-5.6-luna high stableswap-ng-vyper fe88665: 4.04 - - -gpt-5.6-luna high stableswap-ng-vyper 912869e: 5.47 - - -gpt-5.6-luna high stableswap-ng-vyper 1b8fb0a: 5.03 - - -gpt-5.6-luna high stableswap-ng-vyper 4122b3b: unavailable (pricing-unavailable) - - -n/a 4122b3b -gpt-5.6-luna high stableswap-ng-vyper 06db802: 4.34 - - -gpt-5.6-luna high stableswap-ng-vyper c011d7d: 5.81 - - -gpt-5.6-luna high stableswap-ng-vyper 9ea6ca2: 5.31 - - -gpt-5.6-luna high stableswap-ng-vyper 6a3a7a9: 4.46 - - -gpt-5.6-luna high stableswap-ng-vyper 49dbae3: 5.46 - - -gpt-5.6-luna high stableswap-ng-vyper 5302ff0: 6.29 - - -gpt-5.6-luna high stableswap-ng-vyper 82879ed: 5.46 - - -gpt-5.6-luna high stableswap-ng-vyper 0577d17: 5.22 - - -gpt-5.6-luna high stableswap-ng-vyper 381ba55: 5.89 - - -gpt-5.6-luna high stableswap-ng-vyper 386727c: 5.77 - - - -gpt-5.6-luna high venus-isolated-pools-hardhat 953a397: 17.3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2051b64: 24.1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 23ef040: 16.79 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8483637: 19.79 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e03c176: 25.69 - - -gpt-5.6-luna high venus-isolated-pools-hardhat da5d191: 18.79 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 24b90c4: 26.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5cfa918: 18.64 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bdf2482: 22.67 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed26197: 19.24 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3ea3059: 32.63 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 63becee: 3.67 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bf9f729: 4.01 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8aac39f: 3.22 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 80350a9: 2.94 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4928cbe: 6.65 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e685be8: 3.55 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 10f5608: 4.42 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 04cfef8: 5.13 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fa8d039: 3.97 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 89e05a7: 2.71 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0e073cf: 3.53 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 09deffa: 4.37 - - -gpt-5.6-luna high venus-isolated-pools-hardhat a4b3c97: 5.26 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650dc02: 6.22 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4c59002: 2.97 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fd079e3: 2.51 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e750760: 3.73 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5b7c0c0: 3.88 - - -gpt-5.6-luna high venus-isolated-pools-hardhat b22f23c: 3.08 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2d47e73: 2.73 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5db8c6f: 2.97 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650a87a: 4.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat cb672c6: 2.91 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6d1db74: 4.01 - - -gpt-5.6-luna high venus-isolated-pools-hardhat dd874e4: unavailable (pricing-unavailable) - - -n/a dd874e4 -gpt-5.6-luna high venus-isolated-pools-hardhat 149db7d: 3.24 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed59513: 4.67 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 12d716e: 5 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1775fec: 4.28 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3663089: 3.94 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0f5117f: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 87442ee: 5.66 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ac03a8d: 4.56 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2a2efb2: unavailable (pricing-unavailable) - - -n/a 2a2efb2 -gpt-5.6-luna high venus-isolated-pools-hardhat fe88665: 6.25 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 912869e: 7.53 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1b8fb0a: unavailable (pricing-unavailable) - - -n/a 1b8fb0a -gpt-5.6-luna high venus-isolated-pools-hardhat 4122b3b: unavailable (pricing-unavailable) - - -n/a 4122b3b -gpt-5.6-luna high venus-isolated-pools-hardhat 06db802: 5.41 - - -gpt-5.6-luna high venus-isolated-pools-hardhat c011d7d: unavailable (pricing-unavailable) - - -n/a c011d7d -gpt-5.6-luna high venus-isolated-pools-hardhat 9ea6ca2: 6.4 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6a3a7a9: 5.17 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 49dbae3: 5.77 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5302ff0: 5.86 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 82879ed: 7.48 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0577d17: 6.19 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 381ba55: 6.14 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 386727c: 6.7 - - - -gpt-5.6-luna high very-liquid-vaults-foundry 953a397: 24.24 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2051b64: 33.17 - - -gpt-5.6-luna high very-liquid-vaults-foundry 23ef040: 29.34 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8483637: 28.99 - - -gpt-5.6-luna high very-liquid-vaults-foundry e03c176: 31.12 - - -gpt-5.6-luna high very-liquid-vaults-foundry da5d191: 33.01 - - -gpt-5.6-luna high very-liquid-vaults-foundry 24b90c4: 23.88 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5cfa918: 25.27 - - -gpt-5.6-luna high very-liquid-vaults-foundry bdf2482: 20 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed26197: 21.27 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3ea3059: 26.59 - - -gpt-5.6-luna high very-liquid-vaults-foundry 63becee: 7.64 - - -gpt-5.6-luna high very-liquid-vaults-foundry bf9f729: 6.63 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8aac39f: 5.3 - - -gpt-5.6-luna high very-liquid-vaults-foundry 80350a9: 5.32 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4928cbe: 4.52 - - -gpt-5.6-luna high very-liquid-vaults-foundry e685be8: 4.02 - - -gpt-5.6-luna high very-liquid-vaults-foundry 10f5608: 4.84 - - -gpt-5.6-luna high very-liquid-vaults-foundry 04cfef8: 4.94 - - -gpt-5.6-luna high very-liquid-vaults-foundry fa8d039: 4.03 - - -gpt-5.6-luna high very-liquid-vaults-foundry 89e05a7: 4.75 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0e073cf: 5.85 - - -gpt-5.6-luna high very-liquid-vaults-foundry 09deffa: 3.42 - - -gpt-5.6-luna high very-liquid-vaults-foundry a4b3c97: 4.72 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650dc02: 4.46 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4c59002: 7.77 - - -gpt-5.6-luna high very-liquid-vaults-foundry fd079e3: 4.99 - - -gpt-5.6-luna high very-liquid-vaults-foundry e750760: 3.92 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5b7c0c0: 4.59 - - -gpt-5.6-luna high very-liquid-vaults-foundry b22f23c: 5.02 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2d47e73: 4.81 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5db8c6f: 4.12 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650a87a: 3.93 - - -gpt-5.6-luna high very-liquid-vaults-foundry cb672c6: 5.67 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6d1db74: 3.2 - - -gpt-5.6-luna high very-liquid-vaults-foundry dd874e4: 6.6 - - -gpt-5.6-luna high very-liquid-vaults-foundry 149db7d: 3.8 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed59513: 5.7 - - -gpt-5.6-luna high very-liquid-vaults-foundry 12d716e: 7.8 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1775fec: 6.99 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3663089: 4.84 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0f5117f: 5.81 - - -gpt-5.6-luna high very-liquid-vaults-foundry 87442ee: 4.74 - - -gpt-5.6-luna high very-liquid-vaults-foundry ac03a8d: 3.96 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2a2efb2: 6.65 - - -gpt-5.6-luna high very-liquid-vaults-foundry fe88665: 4.03 - - -gpt-5.6-luna high very-liquid-vaults-foundry 912869e: 6.83 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1b8fb0a: 9.76 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4122b3b: 5.16 - - -gpt-5.6-luna high very-liquid-vaults-foundry 06db802: 8.82 - - -gpt-5.6-luna high very-liquid-vaults-foundry c011d7d: 6.64 - - -gpt-5.6-luna high very-liquid-vaults-foundry 9ea6ca2: 6.21 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6a3a7a9: 5.37 - - -gpt-5.6-luna high very-liquid-vaults-foundry 49dbae3: 4.96 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5302ff0: 5.42 - - -gpt-5.6-luna high very-liquid-vaults-foundry 82879ed: 6.07 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0577d17: 8.28 - - -gpt-5.6-luna high very-liquid-vaults-foundry 381ba55: 7.24 - - -gpt-5.6-luna high very-liquid-vaults-foundry 386727c: 6.18 - - -kimi-k3 max stableswap-ng-vyper bdf2482: unavailable (accounting-unavailable) - - -n/a bdf2482 -kimi-k3 max stableswap-ng-vyper 5771bbc: 5.93 - - -kimi-k3 max venus-isolated-pools-hardhat bdf2482: unavailable (accounting-unavailable) - - -n/a bdf2482 -kimi-k3 max venus-isolated-pools-hardhat 5771bbc: 5.42 - - -kimi-k3 max very-liquid-vaults-foundry bdf2482: unavailable (accounting-unavailable) - - -n/a bdf2482 -kimi-k3 max very-liquid-vaults-foundry 5771bbc: 5.32 - - -953a397 -2026-07-21 -2051b64 -2026-07-22 -23ef040 -2026-07-22 -8483637 -2026-07-22 -e03c176 -2026-07-23 -da5d191 -2026-07-23 -24b90c4 -2026-07-30 -5cfa918 -2026-07-30 -bdf2482 -2026-07-30 -bdf2482 -2026-07-30 -ed26197 -2026-07-30 -3ea3059 -2026-07-30 -63becee -2026-07-31 -bf9f729 -2026-07-31 -8aac39f -2026-07-31 -80350a9 -2026-07-31 -4928cbe -2026-07-31 -09277b8 -2026-08-01 -e685be8 -2026-08-02 -10f5608 -2026-08-03 -5771bbc -2026-08-03 -04cfef8 -2026-08-03 -fa8d039 -2026-08-04 -89e05a7 -2026-08-04 -0e073cf -2026-08-04 -09deffa -2026-08-05 -a4b3c97 -2026-08-05 -650dc02 -2026-08-05 -4c59002 -2026-08-05 -fd079e3 -2026-08-05 -e750760 -2026-08-05 -5b7c0c0 -2026-08-05 -b22f23c -2026-08-05 -2d47e73 -2026-08-07 -5db8c6f -2026-08-07 -650a87a -2026-08-08 -cb672c6 -2026-08-09 -6d1db74 -2026-08-09 -6d1db74 -2026-08-09 -dd874e4 -2026-08-09 -149db7d -2026-08-09 -ed59513 -2026-08-09 -12d716e -2026-08-09 -1775fec -2026-08-09 -3663089 -2026-08-10 -0f5117f -2026-08-10 -87442ee -2026-08-10 -ac03a8d -2026-08-10 -2a2efb2 -2026-08-10 -fe88665 -2026-08-11 -912869e -2026-08-11 -1b8fb0a -2026-08-11 -4122b3b -2026-08-11 -06db802 -2026-08-11 -c011d7d -2026-08-11 -9ea6ca2 -2026-08-11 -6a3a7a9 -2026-08-12 -49dbae3 -2026-09-05 -5302ff0 -2026-09-08 -82879ed -2026-09-10 -0577d17 -2026-09-10 -381ba55 -2026-09-11 -386727c -2026-09-11 - -deepseek-v4-flash max stableswap-ng-vyper - -deepseek-v4-flash max venus-isolated-pools-hardhat - -deepseek-v4-flash max very-liquid-vaults-foundry - -deepseek-v4-pro max stableswap-ng-vyper - -deepseek-v4-pro max venus-isolated-pools-hardhat - -deepseek-v4-pro max very-liquid-vaults-foundry - -gpt-5.6-luna high stableswap-ng-vyper - -gpt-5.6-luna high venus-isolated-pools-hardhat - -gpt-5.6-luna high very-liquid-vaults-foundry - -kimi-k3 max stableswap-ng-vyper - -kimi-k3 max venus-isolated-pools-hardhat - -kimi-k3 max very-liquid-vaults-foundry - diff --git a/docs/assets/eval-history/cumulative-unique-true-positives.svg b/docs/assets/eval-history/cumulative-unique-true-positives.svg deleted file mode 100644 index e0fa96e99..000000000 --- a/docs/assets/eval-history/cumulative-unique-true-positives.svg +++ /dev/null @@ -1,771 +0,0 @@ - - -Cumulative unique true positives -Cumulative unique true positives by candidate commit and benchmark target - -Cumulative unique true positives -ultrafuzz-bench · smoke -Each line tracks one benchmark target across evenly spaced candidate-run columns. - - - -0 - -1 - -2 - -3 - -4 -ultrafuzz-bench smoke cohort-cf3da608 policy-b0f7d6d0 - -cohort-cf3da608 -ultrafuzz-bench smoke cohort-a98bf547 policy-33b8d296 - -cohort-a98bf547 -ultrafuzz-bench smoke cohort-7d061c80 policy-33b8d296 - -cohort-7d061c80 -ultrafuzz-bench smoke cohort-b05ba222 policy-33b8d296 - -cohort-b05ba222 -ultrafuzz-bench smoke cohort-c0e17843 policy-33b8d296 - -cohort-c0e17843 -ultrafuzz-bench smoke cohort-d49aa3e2 policy-33b8d296 - -cohort-d49aa3e2 -ultrafuzz-bench smoke cohort-5d6ac3fd policy-fbbf61e0 - -cohort-5d6ac3fd -ultrafuzz-bench smoke cohort-21f261d4 policy-dfea2bb8 - -cohort-21f261d4 -ultrafuzz-bench smoke cohort-f3e78e3d policy-a7379336 - -cohort-f3e78e3d -deepseek-v4-flash max stableswap-ng-vyper 6d1db74: 1 - - -deepseek-v4-flash max venus-isolated-pools-hardhat 6d1db74: 1 - - -deepseek-v4-flash max very-liquid-vaults-foundry 6d1db74: 2 - - -deepseek-v4-pro max stableswap-ng-vyper 09277b8: 0 - - -deepseek-v4-pro max venus-isolated-pools-hardhat 09277b8: 3 - - -deepseek-v4-pro max very-liquid-vaults-foundry 09277b8: 0 - - - -gpt-5.6-luna high stableswap-ng-vyper 953a397: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 2051b64: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 23ef040: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 8483637: 1 - - -gpt-5.6-luna high stableswap-ng-vyper e03c176: 1 - - -gpt-5.6-luna high stableswap-ng-vyper da5d191: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 24b90c4: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 5cfa918: 0 - - -gpt-5.6-luna high stableswap-ng-vyper bdf2482: 1 - - -gpt-5.6-luna high stableswap-ng-vyper ed26197: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 3ea3059: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 63becee: 1 - - -gpt-5.6-luna high stableswap-ng-vyper bf9f729: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 8aac39f: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 80350a9: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 4928cbe: 0 - - -gpt-5.6-luna high stableswap-ng-vyper e685be8: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 10f5608: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 04cfef8: 1 - - -gpt-5.6-luna high stableswap-ng-vyper fa8d039: 0 - - -gpt-5.6-luna high stableswap-ng-vyper 89e05a7: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 0e073cf: 0 - - -gpt-5.6-luna high stableswap-ng-vyper 09deffa: 0 - - -gpt-5.6-luna high stableswap-ng-vyper a4b3c97: 0 - - -gpt-5.6-luna high stableswap-ng-vyper 650dc02: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 4c59002: 0 - - -gpt-5.6-luna high stableswap-ng-vyper fd079e3: 1 - - -gpt-5.6-luna high stableswap-ng-vyper e750760: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 5b7c0c0: 0 - - -gpt-5.6-luna high stableswap-ng-vyper b22f23c: 0 - - -gpt-5.6-luna high stableswap-ng-vyper 2d47e73: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 5db8c6f: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 650a87a: 1 - - -gpt-5.6-luna high stableswap-ng-vyper cb672c6: 0 - - -gpt-5.6-luna high stableswap-ng-vyper 6d1db74: 1 - - -gpt-5.6-luna high stableswap-ng-vyper dd874e4: 0 - - -gpt-5.6-luna high stableswap-ng-vyper 149db7d: 0 - - -gpt-5.6-luna high stableswap-ng-vyper ed59513: 0 - - -gpt-5.6-luna high stableswap-ng-vyper 12d716e: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 1775fec: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 3663089: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 0f5117f: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 87442ee: 1 - - -gpt-5.6-luna high stableswap-ng-vyper ac03a8d: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 2a2efb2: 1 - - -gpt-5.6-luna high stableswap-ng-vyper fe88665: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 912869e: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 1b8fb0a: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 4122b3b: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 06db802: 1 - - -gpt-5.6-luna high stableswap-ng-vyper c011d7d: 0 - - -gpt-5.6-luna high stableswap-ng-vyper 9ea6ca2: 0 - - -gpt-5.6-luna high stableswap-ng-vyper 6a3a7a9: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 49dbae3: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 5302ff0: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 82879ed: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 0577d17: 1 - - -gpt-5.6-luna high stableswap-ng-vyper 381ba55: 2 - - -gpt-5.6-luna high stableswap-ng-vyper 386727c: 1 - - - -gpt-5.6-luna high venus-isolated-pools-hardhat 953a397: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2051b64: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 23ef040: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8483637: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e03c176: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat da5d191: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 24b90c4: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5cfa918: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bdf2482: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed26197: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3ea3059: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 63becee: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bf9f729: 0 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8aac39f: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 80350a9: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4928cbe: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e685be8: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 10f5608: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 04cfef8: 0 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fa8d039: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 89e05a7: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0e073cf: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 09deffa: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat a4b3c97: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650dc02: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4c59002: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fd079e3: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e750760: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5b7c0c0: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat b22f23c: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2d47e73: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5db8c6f: 4 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650a87a: 0 - - -gpt-5.6-luna high venus-isolated-pools-hardhat cb672c6: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6d1db74: 0 - - -gpt-5.6-luna high venus-isolated-pools-hardhat dd874e4: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 149db7d: 4 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed59513: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 12d716e: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1775fec: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3663089: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0f5117f: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 87442ee: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ac03a8d: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2a2efb2: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fe88665: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 912869e: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1b8fb0a: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4122b3b: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 06db802: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat c011d7d: 4 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 9ea6ca2: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6a3a7a9: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 49dbae3: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5302ff0: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 82879ed: 2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0577d17: 3 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 381ba55: 1 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 386727c: 1 - - - -gpt-5.6-luna high very-liquid-vaults-foundry 953a397: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2051b64: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 23ef040: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8483637: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry e03c176: 0 - - -gpt-5.6-luna high very-liquid-vaults-foundry da5d191: 3 - - -gpt-5.6-luna high very-liquid-vaults-foundry 24b90c4: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5cfa918: 0 - - -gpt-5.6-luna high very-liquid-vaults-foundry bdf2482: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed26197: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3ea3059: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 63becee: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry bf9f729: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8aac39f: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 80350a9: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4928cbe: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry e685be8: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 10f5608: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 04cfef8: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry fa8d039: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 89e05a7: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0e073cf: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 09deffa: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry a4b3c97: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650dc02: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4c59002: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry fd079e3: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry e750760: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5b7c0c0: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry b22f23c: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2d47e73: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5db8c6f: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650a87a: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry cb672c6: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6d1db74: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry dd874e4: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 149db7d: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed59513: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 12d716e: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1775fec: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3663089: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0f5117f: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 87442ee: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry ac03a8d: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2a2efb2: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry fe88665: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 912869e: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1b8fb0a: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4122b3b: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 06db802: 0 - - -gpt-5.6-luna high very-liquid-vaults-foundry c011d7d: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 9ea6ca2: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6a3a7a9: 2 - - -gpt-5.6-luna high very-liquid-vaults-foundry 49dbae3: 0 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5302ff0: 0 - - -gpt-5.6-luna high very-liquid-vaults-foundry 82879ed: 1 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0577d17: 0 - - -gpt-5.6-luna high very-liquid-vaults-foundry 381ba55: 0 - - -gpt-5.6-luna high very-liquid-vaults-foundry 386727c: 0 - - - -kimi-k3 max stableswap-ng-vyper bdf2482: 2 - - -kimi-k3 max stableswap-ng-vyper 5771bbc: 1 - - - -kimi-k3 max venus-isolated-pools-hardhat bdf2482: 3 - - -kimi-k3 max venus-isolated-pools-hardhat 5771bbc: 4 - - - -kimi-k3 max very-liquid-vaults-foundry bdf2482: 2 - - -kimi-k3 max very-liquid-vaults-foundry 5771bbc: 1 - - -953a397 -2026-07-21 -2051b64 -2026-07-22 -23ef040 -2026-07-22 -8483637 -2026-07-22 -e03c176 -2026-07-23 -da5d191 -2026-07-23 -24b90c4 -2026-07-30 -5cfa918 -2026-07-30 -bdf2482 -2026-07-30 -bdf2482 -2026-07-30 -ed26197 -2026-07-30 -3ea3059 -2026-07-30 -63becee -2026-07-31 -bf9f729 -2026-07-31 -8aac39f -2026-07-31 -80350a9 -2026-07-31 -4928cbe -2026-07-31 -09277b8 -2026-08-01 -e685be8 -2026-08-02 -10f5608 -2026-08-03 -5771bbc -2026-08-03 -04cfef8 -2026-08-03 -fa8d039 -2026-08-04 -89e05a7 -2026-08-04 -0e073cf -2026-08-04 -09deffa -2026-08-05 -a4b3c97 -2026-08-05 -650dc02 -2026-08-05 -4c59002 -2026-08-05 -fd079e3 -2026-08-05 -e750760 -2026-08-05 -5b7c0c0 -2026-08-05 -b22f23c -2026-08-05 -2d47e73 -2026-08-07 -5db8c6f -2026-08-07 -650a87a -2026-08-08 -cb672c6 -2026-08-09 -6d1db74 -2026-08-09 -6d1db74 -2026-08-09 -dd874e4 -2026-08-09 -149db7d -2026-08-09 -ed59513 -2026-08-09 -12d716e -2026-08-09 -1775fec -2026-08-09 -3663089 -2026-08-10 -0f5117f -2026-08-10 -87442ee -2026-08-10 -ac03a8d -2026-08-10 -2a2efb2 -2026-08-10 -fe88665 -2026-08-11 -912869e -2026-08-11 -1b8fb0a -2026-08-11 -4122b3b -2026-08-11 -06db802 -2026-08-11 -c011d7d -2026-08-11 -9ea6ca2 -2026-08-11 -6a3a7a9 -2026-08-12 -49dbae3 -2026-09-05 -5302ff0 -2026-09-08 -82879ed -2026-09-10 -0577d17 -2026-09-10 -381ba55 -2026-09-11 -386727c -2026-09-11 - -deepseek-v4-flash max stableswap-ng-vyper - -deepseek-v4-flash max venus-isolated-pools-hardhat - -deepseek-v4-flash max very-liquid-vaults-foundry - -deepseek-v4-pro max stableswap-ng-vyper - -deepseek-v4-pro max venus-isolated-pools-hardhat - -deepseek-v4-pro max very-liquid-vaults-foundry - -gpt-5.6-luna high stableswap-ng-vyper - -gpt-5.6-luna high venus-isolated-pools-hardhat - -gpt-5.6-luna high very-liquid-vaults-foundry - -kimi-k3 max stableswap-ng-vyper - -kimi-k3 max venus-isolated-pools-hardhat - -kimi-k3 max very-liquid-vaults-foundry - diff --git a/docs/assets/eval-history/f1.svg b/docs/assets/eval-history/f1.svg deleted file mode 100644 index 1b1c4d8b4..000000000 --- a/docs/assets/eval-history/f1.svg +++ /dev/null @@ -1,771 +0,0 @@ - - -F1 -F1 by candidate commit and benchmark target - -F1 -ultrafuzz-bench · smoke -Each line tracks one benchmark target across evenly spaced candidate-run columns. - - - -0.00 - -0.25 - -0.50 - -0.75 - -1.00 -ultrafuzz-bench smoke cohort-cf3da608 policy-b0f7d6d0 - -cohort-cf3da608 -ultrafuzz-bench smoke cohort-a98bf547 policy-33b8d296 - -cohort-a98bf547 -ultrafuzz-bench smoke cohort-7d061c80 policy-33b8d296 - -cohort-7d061c80 -ultrafuzz-bench smoke cohort-b05ba222 policy-33b8d296 - -cohort-b05ba222 -ultrafuzz-bench smoke cohort-c0e17843 policy-33b8d296 - -cohort-c0e17843 -ultrafuzz-bench smoke cohort-d49aa3e2 policy-33b8d296 - -cohort-d49aa3e2 -ultrafuzz-bench smoke cohort-5d6ac3fd policy-fbbf61e0 - -cohort-5d6ac3fd -ultrafuzz-bench smoke cohort-21f261d4 policy-dfea2bb8 - -cohort-21f261d4 -ultrafuzz-bench smoke cohort-f3e78e3d policy-a7379336 - -cohort-f3e78e3d -deepseek-v4-flash max stableswap-ng-vyper 6d1db74: 0.25 - - -deepseek-v4-flash max venus-isolated-pools-hardhat 6d1db74: 0.11 - - -deepseek-v4-flash max very-liquid-vaults-foundry 6d1db74: 0.44 - - -deepseek-v4-pro max stableswap-ng-vyper 09277b8: 0.00 - - -deepseek-v4-pro max venus-isolated-pools-hardhat 09277b8: 0.30 - - -deepseek-v4-pro max very-liquid-vaults-foundry 09277b8: 0.00 - - - -gpt-5.6-luna high stableswap-ng-vyper 953a397: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 2051b64: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 23ef040: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 8483637: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper e03c176: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper da5d191: 0.44 - - -gpt-5.6-luna high stableswap-ng-vyper 24b90c4: 0.44 - - -gpt-5.6-luna high stableswap-ng-vyper 5cfa918: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper bdf2482: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper ed26197: 0.22 - - -gpt-5.6-luna high stableswap-ng-vyper 3ea3059: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 63becee: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper bf9f729: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 8aac39f: 0.40 - - -gpt-5.6-luna high stableswap-ng-vyper 80350a9: 0.36 - - -gpt-5.6-luna high stableswap-ng-vyper 4928cbe: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper e685be8: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 10f5608: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 04cfef8: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper fa8d039: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 89e05a7: 0.44 - - -gpt-5.6-luna high stableswap-ng-vyper 0e073cf: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 09deffa: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper a4b3c97: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 650dc02: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 4c59002: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper fd079e3: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper e750760: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 5b7c0c0: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper b22f23c: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 2d47e73: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 5db8c6f: 0.44 - - -gpt-5.6-luna high stableswap-ng-vyper 650a87a: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper cb672c6: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 6d1db74: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper dd874e4: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 149db7d: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper ed59513: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 12d716e: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 1775fec: 0.44 - - -gpt-5.6-luna high stableswap-ng-vyper 3663089: 0.44 - - -gpt-5.6-luna high stableswap-ng-vyper 0f5117f: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 87442ee: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper ac03a8d: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 2a2efb2: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper fe88665: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 912869e: 0.44 - - -gpt-5.6-luna high stableswap-ng-vyper 1b8fb0a: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 4122b3b: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 06db802: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper c011d7d: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 9ea6ca2: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 6a3a7a9: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 49dbae3: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 5302ff0: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 82879ed: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 0577d17: 0.25 - - -gpt-5.6-luna high stableswap-ng-vyper 381ba55: 0.44 - - -gpt-5.6-luna high stableswap-ng-vyper 386727c: 0.25 - - - -gpt-5.6-luna high venus-isolated-pools-hardhat 953a397: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2051b64: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 23ef040: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8483637: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e03c176: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat da5d191: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 24b90c4: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5cfa918: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bdf2482: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed26197: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3ea3059: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 63becee: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bf9f729: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8aac39f: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 80350a9: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4928cbe: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e685be8: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 10f5608: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 04cfef8: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fa8d039: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 89e05a7: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0e073cf: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 09deffa: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat a4b3c97: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650dc02: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4c59002: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fd079e3: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e750760: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5b7c0c0: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat b22f23c: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2d47e73: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5db8c6f: 0.38 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650a87a: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat cb672c6: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6d1db74: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat dd874e4: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 149db7d: 0.38 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed59513: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 12d716e: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1775fec: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3663089: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0f5117f: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 87442ee: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ac03a8d: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2a2efb2: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fe88665: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 912869e: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1b8fb0a: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4122b3b: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 06db802: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat c011d7d: 0.38 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 9ea6ca2: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6a3a7a9: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 49dbae3: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5302ff0: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 82879ed: 0.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0577d17: 0.30 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 381ba55: 0.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 386727c: 0.11 - - - -gpt-5.6-luna high very-liquid-vaults-foundry 953a397: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2051b64: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 23ef040: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8483637: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry e03c176: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry da5d191: 0.60 - - -gpt-5.6-luna high very-liquid-vaults-foundry 24b90c4: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5cfa918: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry bdf2482: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed26197: 0.40 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3ea3059: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 63becee: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry bf9f729: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8aac39f: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 80350a9: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4928cbe: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry e685be8: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 10f5608: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 04cfef8: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry fa8d039: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 89e05a7: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0e073cf: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 09deffa: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry a4b3c97: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650dc02: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4c59002: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry fd079e3: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry e750760: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5b7c0c0: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry b22f23c: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2d47e73: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5db8c6f: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650a87a: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry cb672c6: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6d1db74: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry dd874e4: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 149db7d: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed59513: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 12d716e: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1775fec: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3663089: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0f5117f: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 87442ee: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry ac03a8d: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2a2efb2: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry fe88665: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 912869e: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1b8fb0a: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4122b3b: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 06db802: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry c011d7d: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 9ea6ca2: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6a3a7a9: 0.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 49dbae3: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5302ff0: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 82879ed: 0.25 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0577d17: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 381ba55: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 386727c: 0.00 - - - -kimi-k3 max stableswap-ng-vyper bdf2482: 0.44 - - -kimi-k3 max stableswap-ng-vyper 5771bbc: 0.25 - - - -kimi-k3 max venus-isolated-pools-hardhat bdf2482: 0.30 - - -kimi-k3 max venus-isolated-pools-hardhat 5771bbc: 0.38 - - - -kimi-k3 max very-liquid-vaults-foundry bdf2482: 0.44 - - -kimi-k3 max very-liquid-vaults-foundry 5771bbc: 0.25 - - -953a397 -2026-07-21 -2051b64 -2026-07-22 -23ef040 -2026-07-22 -8483637 -2026-07-22 -e03c176 -2026-07-23 -da5d191 -2026-07-23 -24b90c4 -2026-07-30 -5cfa918 -2026-07-30 -bdf2482 -2026-07-30 -bdf2482 -2026-07-30 -ed26197 -2026-07-30 -3ea3059 -2026-07-30 -63becee -2026-07-31 -bf9f729 -2026-07-31 -8aac39f -2026-07-31 -80350a9 -2026-07-31 -4928cbe -2026-07-31 -09277b8 -2026-08-01 -e685be8 -2026-08-02 -10f5608 -2026-08-03 -5771bbc -2026-08-03 -04cfef8 -2026-08-03 -fa8d039 -2026-08-04 -89e05a7 -2026-08-04 -0e073cf -2026-08-04 -09deffa -2026-08-05 -a4b3c97 -2026-08-05 -650dc02 -2026-08-05 -4c59002 -2026-08-05 -fd079e3 -2026-08-05 -e750760 -2026-08-05 -5b7c0c0 -2026-08-05 -b22f23c -2026-08-05 -2d47e73 -2026-08-07 -5db8c6f -2026-08-07 -650a87a -2026-08-08 -cb672c6 -2026-08-09 -6d1db74 -2026-08-09 -6d1db74 -2026-08-09 -dd874e4 -2026-08-09 -149db7d -2026-08-09 -ed59513 -2026-08-09 -12d716e -2026-08-09 -1775fec -2026-08-09 -3663089 -2026-08-10 -0f5117f -2026-08-10 -87442ee -2026-08-10 -ac03a8d -2026-08-10 -2a2efb2 -2026-08-10 -fe88665 -2026-08-11 -912869e -2026-08-11 -1b8fb0a -2026-08-11 -4122b3b -2026-08-11 -06db802 -2026-08-11 -c011d7d -2026-08-11 -9ea6ca2 -2026-08-11 -6a3a7a9 -2026-08-12 -49dbae3 -2026-09-05 -5302ff0 -2026-09-08 -82879ed -2026-09-10 -0577d17 -2026-09-10 -381ba55 -2026-09-11 -386727c -2026-09-11 - -deepseek-v4-flash max stableswap-ng-vyper - -deepseek-v4-flash max venus-isolated-pools-hardhat - -deepseek-v4-flash max very-liquid-vaults-foundry - -deepseek-v4-pro max stableswap-ng-vyper - -deepseek-v4-pro max venus-isolated-pools-hardhat - -deepseek-v4-pro max very-liquid-vaults-foundry - -gpt-5.6-luna high stableswap-ng-vyper - -gpt-5.6-luna high venus-isolated-pools-hardhat - -gpt-5.6-luna high very-liquid-vaults-foundry - -kimi-k3 max stableswap-ng-vyper - -kimi-k3 max venus-isolated-pools-hardhat - -kimi-k3 max very-liquid-vaults-foundry - diff --git a/docs/assets/eval-history/precision.svg b/docs/assets/eval-history/precision.svg deleted file mode 100644 index 21b75c99f..000000000 --- a/docs/assets/eval-history/precision.svg +++ /dev/null @@ -1,771 +0,0 @@ - - -Precision -Precision by candidate commit and benchmark target - -Precision -ultrafuzz-bench · smoke -Each line tracks one benchmark target across evenly spaced candidate-run columns. - - - -0.00 - -0.25 - -0.50 - -0.75 - -1.00 -ultrafuzz-bench smoke cohort-cf3da608 policy-b0f7d6d0 - -cohort-cf3da608 -ultrafuzz-bench smoke cohort-a98bf547 policy-33b8d296 - -cohort-a98bf547 -ultrafuzz-bench smoke cohort-7d061c80 policy-33b8d296 - -cohort-7d061c80 -ultrafuzz-bench smoke cohort-b05ba222 policy-33b8d296 - -cohort-b05ba222 -ultrafuzz-bench smoke cohort-c0e17843 policy-33b8d296 - -cohort-c0e17843 -ultrafuzz-bench smoke cohort-d49aa3e2 policy-33b8d296 - -cohort-d49aa3e2 -ultrafuzz-bench smoke cohort-5d6ac3fd policy-fbbf61e0 - -cohort-5d6ac3fd -ultrafuzz-bench smoke cohort-21f261d4 policy-dfea2bb8 - -cohort-21f261d4 -ultrafuzz-bench smoke cohort-f3e78e3d policy-a7379336 - -cohort-f3e78e3d -deepseek-v4-flash max stableswap-ng-vyper 6d1db74: 1.00 - - -deepseek-v4-flash max venus-isolated-pools-hardhat 6d1db74: 1.00 - - -deepseek-v4-flash max very-liquid-vaults-foundry 6d1db74: 1.00 - - -deepseek-v4-pro max stableswap-ng-vyper 09277b8: 0.00 - - -deepseek-v4-pro max venus-isolated-pools-hardhat 09277b8: 1.00 - - -deepseek-v4-pro max very-liquid-vaults-foundry 09277b8: 0.00 - - - -gpt-5.6-luna high stableswap-ng-vyper 953a397: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 2051b64: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 23ef040: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 8483637: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper e03c176: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper da5d191: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 24b90c4: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 5cfa918: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper bdf2482: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper ed26197: 0.50 - - -gpt-5.6-luna high stableswap-ng-vyper 3ea3059: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 63becee: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper bf9f729: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 8aac39f: 0.67 - - -gpt-5.6-luna high stableswap-ng-vyper 80350a9: 0.50 - - -gpt-5.6-luna high stableswap-ng-vyper 4928cbe: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper e685be8: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 10f5608: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 04cfef8: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper fa8d039: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 89e05a7: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 0e073cf: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 09deffa: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper a4b3c97: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 650dc02: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 4c59002: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper fd079e3: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper e750760: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 5b7c0c0: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper b22f23c: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 2d47e73: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 5db8c6f: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 650a87a: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper cb672c6: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 6d1db74: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper dd874e4: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 149db7d: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper ed59513: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 12d716e: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 1775fec: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 3663089: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 0f5117f: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 87442ee: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper ac03a8d: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 2a2efb2: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper fe88665: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 912869e: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 1b8fb0a: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 4122b3b: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 06db802: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper c011d7d: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 9ea6ca2: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 6a3a7a9: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 49dbae3: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 5302ff0: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 82879ed: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 0577d17: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 381ba55: 1.00 - - -gpt-5.6-luna high stableswap-ng-vyper 386727c: 1.00 - - - -gpt-5.6-luna high venus-isolated-pools-hardhat 953a397: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2051b64: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 23ef040: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8483637: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e03c176: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat da5d191: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 24b90c4: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5cfa918: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bdf2482: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed26197: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3ea3059: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 63becee: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bf9f729: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8aac39f: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 80350a9: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4928cbe: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e685be8: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 10f5608: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 04cfef8: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fa8d039: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 89e05a7: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0e073cf: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 09deffa: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat a4b3c97: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650dc02: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4c59002: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fd079e3: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e750760: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5b7c0c0: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat b22f23c: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2d47e73: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5db8c6f: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650a87a: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat cb672c6: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6d1db74: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat dd874e4: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 149db7d: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed59513: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 12d716e: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1775fec: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3663089: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0f5117f: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 87442ee: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ac03a8d: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2a2efb2: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fe88665: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 912869e: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1b8fb0a: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4122b3b: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 06db802: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat c011d7d: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 9ea6ca2: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6a3a7a9: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 49dbae3: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5302ff0: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 82879ed: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0577d17: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 381ba55: 1.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 386727c: 1.00 - - - -gpt-5.6-luna high very-liquid-vaults-foundry 953a397: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2051b64: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 23ef040: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8483637: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry e03c176: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry da5d191: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 24b90c4: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5cfa918: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry bdf2482: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed26197: 0.67 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3ea3059: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 63becee: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry bf9f729: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8aac39f: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 80350a9: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4928cbe: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry e685be8: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 10f5608: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 04cfef8: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry fa8d039: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 89e05a7: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0e073cf: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 09deffa: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry a4b3c97: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650dc02: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4c59002: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry fd079e3: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry e750760: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5b7c0c0: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry b22f23c: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2d47e73: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5db8c6f: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650a87a: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry cb672c6: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6d1db74: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry dd874e4: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 149db7d: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed59513: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 12d716e: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1775fec: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3663089: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0f5117f: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 87442ee: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry ac03a8d: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2a2efb2: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry fe88665: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 912869e: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1b8fb0a: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4122b3b: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 06db802: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry c011d7d: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 9ea6ca2: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6a3a7a9: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 49dbae3: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5302ff0: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 82879ed: 1.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0577d17: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 381ba55: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 386727c: 0.00 - - - -kimi-k3 max stableswap-ng-vyper bdf2482: 1.00 - - -kimi-k3 max stableswap-ng-vyper 5771bbc: 1.00 - - - -kimi-k3 max venus-isolated-pools-hardhat bdf2482: 1.00 - - -kimi-k3 max venus-isolated-pools-hardhat 5771bbc: 1.00 - - - -kimi-k3 max very-liquid-vaults-foundry bdf2482: 1.00 - - -kimi-k3 max very-liquid-vaults-foundry 5771bbc: 1.00 - - -953a397 -2026-07-21 -2051b64 -2026-07-22 -23ef040 -2026-07-22 -8483637 -2026-07-22 -e03c176 -2026-07-23 -da5d191 -2026-07-23 -24b90c4 -2026-07-30 -5cfa918 -2026-07-30 -bdf2482 -2026-07-30 -bdf2482 -2026-07-30 -ed26197 -2026-07-30 -3ea3059 -2026-07-30 -63becee -2026-07-31 -bf9f729 -2026-07-31 -8aac39f -2026-07-31 -80350a9 -2026-07-31 -4928cbe -2026-07-31 -09277b8 -2026-08-01 -e685be8 -2026-08-02 -10f5608 -2026-08-03 -5771bbc -2026-08-03 -04cfef8 -2026-08-03 -fa8d039 -2026-08-04 -89e05a7 -2026-08-04 -0e073cf -2026-08-04 -09deffa -2026-08-05 -a4b3c97 -2026-08-05 -650dc02 -2026-08-05 -4c59002 -2026-08-05 -fd079e3 -2026-08-05 -e750760 -2026-08-05 -5b7c0c0 -2026-08-05 -b22f23c -2026-08-05 -2d47e73 -2026-08-07 -5db8c6f -2026-08-07 -650a87a -2026-08-08 -cb672c6 -2026-08-09 -6d1db74 -2026-08-09 -6d1db74 -2026-08-09 -dd874e4 -2026-08-09 -149db7d -2026-08-09 -ed59513 -2026-08-09 -12d716e -2026-08-09 -1775fec -2026-08-09 -3663089 -2026-08-10 -0f5117f -2026-08-10 -87442ee -2026-08-10 -ac03a8d -2026-08-10 -2a2efb2 -2026-08-10 -fe88665 -2026-08-11 -912869e -2026-08-11 -1b8fb0a -2026-08-11 -4122b3b -2026-08-11 -06db802 -2026-08-11 -c011d7d -2026-08-11 -9ea6ca2 -2026-08-11 -6a3a7a9 -2026-08-12 -49dbae3 -2026-09-05 -5302ff0 -2026-09-08 -82879ed -2026-09-10 -0577d17 -2026-09-10 -381ba55 -2026-09-11 -386727c -2026-09-11 - -deepseek-v4-flash max stableswap-ng-vyper - -deepseek-v4-flash max venus-isolated-pools-hardhat - -deepseek-v4-flash max very-liquid-vaults-foundry - -deepseek-v4-pro max stableswap-ng-vyper - -deepseek-v4-pro max venus-isolated-pools-hardhat - -deepseek-v4-pro max very-liquid-vaults-foundry - -gpt-5.6-luna high stableswap-ng-vyper - -gpt-5.6-luna high venus-isolated-pools-hardhat - -gpt-5.6-luna high very-liquid-vaults-foundry - -kimi-k3 max stableswap-ng-vyper - -kimi-k3 max venus-isolated-pools-hardhat - -kimi-k3 max very-liquid-vaults-foundry - diff --git a/docs/assets/eval-history/recall.svg b/docs/assets/eval-history/recall.svg deleted file mode 100644 index db500276c..000000000 --- a/docs/assets/eval-history/recall.svg +++ /dev/null @@ -1,771 +0,0 @@ - - -Recall -Recall by candidate commit and benchmark target - -Recall -ultrafuzz-bench · smoke -Each line tracks one benchmark target across evenly spaced candidate-run columns. - - - -0.00 - -0.25 - -0.50 - -0.75 - -1.00 -ultrafuzz-bench smoke cohort-cf3da608 policy-b0f7d6d0 - -cohort-cf3da608 -ultrafuzz-bench smoke cohort-a98bf547 policy-33b8d296 - -cohort-a98bf547 -ultrafuzz-bench smoke cohort-7d061c80 policy-33b8d296 - -cohort-7d061c80 -ultrafuzz-bench smoke cohort-b05ba222 policy-33b8d296 - -cohort-b05ba222 -ultrafuzz-bench smoke cohort-c0e17843 policy-33b8d296 - -cohort-c0e17843 -ultrafuzz-bench smoke cohort-d49aa3e2 policy-33b8d296 - -cohort-d49aa3e2 -ultrafuzz-bench smoke cohort-5d6ac3fd policy-fbbf61e0 - -cohort-5d6ac3fd -ultrafuzz-bench smoke cohort-21f261d4 policy-dfea2bb8 - -cohort-21f261d4 -ultrafuzz-bench smoke cohort-f3e78e3d policy-a7379336 - -cohort-f3e78e3d -deepseek-v4-flash max stableswap-ng-vyper 6d1db74: 0.14 - - -deepseek-v4-flash max venus-isolated-pools-hardhat 6d1db74: 0.06 - - -deepseek-v4-flash max very-liquid-vaults-foundry 6d1db74: 0.29 - - -deepseek-v4-pro max stableswap-ng-vyper 09277b8: 0.00 - - -deepseek-v4-pro max venus-isolated-pools-hardhat 09277b8: 0.18 - - -deepseek-v4-pro max very-liquid-vaults-foundry 09277b8: 0.00 - - - -gpt-5.6-luna high stableswap-ng-vyper 953a397: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 2051b64: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 23ef040: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 8483637: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper e03c176: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper da5d191: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 24b90c4: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 5cfa918: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper bdf2482: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper ed26197: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 3ea3059: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 63becee: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper bf9f729: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 8aac39f: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 80350a9: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 4928cbe: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper e685be8: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 10f5608: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 04cfef8: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper fa8d039: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 89e05a7: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 0e073cf: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 09deffa: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper a4b3c97: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 650dc02: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 4c59002: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper fd079e3: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper e750760: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 5b7c0c0: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper b22f23c: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 2d47e73: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 5db8c6f: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 650a87a: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper cb672c6: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 6d1db74: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper dd874e4: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 149db7d: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper ed59513: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 12d716e: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 1775fec: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 3663089: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 0f5117f: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 87442ee: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper ac03a8d: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 2a2efb2: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper fe88665: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 912869e: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 1b8fb0a: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 4122b3b: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 06db802: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper c011d7d: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 9ea6ca2: 0.00 - - -gpt-5.6-luna high stableswap-ng-vyper 6a3a7a9: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 49dbae3: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 5302ff0: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 82879ed: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 0577d17: 0.14 - - -gpt-5.6-luna high stableswap-ng-vyper 381ba55: 0.29 - - -gpt-5.6-luna high stableswap-ng-vyper 386727c: 0.14 - - - -gpt-5.6-luna high venus-isolated-pools-hardhat 953a397: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2051b64: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 23ef040: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8483637: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e03c176: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat da5d191: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 24b90c4: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5cfa918: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bdf2482: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed26197: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3ea3059: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 63becee: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bf9f729: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8aac39f: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 80350a9: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4928cbe: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e685be8: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 10f5608: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 04cfef8: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fa8d039: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 89e05a7: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0e073cf: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 09deffa: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat a4b3c97: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650dc02: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4c59002: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fd079e3: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e750760: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5b7c0c0: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat b22f23c: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2d47e73: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5db8c6f: 0.24 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650a87a: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat cb672c6: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6d1db74: 0.00 - - -gpt-5.6-luna high venus-isolated-pools-hardhat dd874e4: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 149db7d: 0.24 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed59513: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 12d716e: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1775fec: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3663089: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0f5117f: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 87442ee: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ac03a8d: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2a2efb2: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fe88665: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 912869e: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1b8fb0a: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4122b3b: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 06db802: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat c011d7d: 0.24 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 9ea6ca2: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6a3a7a9: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 49dbae3: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5302ff0: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 82879ed: 0.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0577d17: 0.18 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 381ba55: 0.06 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 386727c: 0.06 - - - -gpt-5.6-luna high very-liquid-vaults-foundry 953a397: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2051b64: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 23ef040: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8483637: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry e03c176: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry da5d191: 0.43 - - -gpt-5.6-luna high very-liquid-vaults-foundry 24b90c4: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5cfa918: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry bdf2482: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed26197: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3ea3059: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 63becee: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry bf9f729: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8aac39f: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 80350a9: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4928cbe: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry e685be8: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 10f5608: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 04cfef8: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry fa8d039: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 89e05a7: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0e073cf: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 09deffa: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry a4b3c97: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650dc02: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4c59002: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry fd079e3: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry e750760: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5b7c0c0: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry b22f23c: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2d47e73: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5db8c6f: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650a87a: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry cb672c6: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6d1db74: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry dd874e4: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 149db7d: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed59513: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 12d716e: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1775fec: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3663089: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0f5117f: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 87442ee: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry ac03a8d: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2a2efb2: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry fe88665: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 912869e: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1b8fb0a: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4122b3b: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 06db802: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry c011d7d: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 9ea6ca2: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6a3a7a9: 0.29 - - -gpt-5.6-luna high very-liquid-vaults-foundry 49dbae3: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5302ff0: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 82879ed: 0.14 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0577d17: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 381ba55: 0.00 - - -gpt-5.6-luna high very-liquid-vaults-foundry 386727c: 0.00 - - - -kimi-k3 max stableswap-ng-vyper bdf2482: 0.29 - - -kimi-k3 max stableswap-ng-vyper 5771bbc: 0.14 - - - -kimi-k3 max venus-isolated-pools-hardhat bdf2482: 0.18 - - -kimi-k3 max venus-isolated-pools-hardhat 5771bbc: 0.24 - - - -kimi-k3 max very-liquid-vaults-foundry bdf2482: 0.29 - - -kimi-k3 max very-liquid-vaults-foundry 5771bbc: 0.14 - - -953a397 -2026-07-21 -2051b64 -2026-07-22 -23ef040 -2026-07-22 -8483637 -2026-07-22 -e03c176 -2026-07-23 -da5d191 -2026-07-23 -24b90c4 -2026-07-30 -5cfa918 -2026-07-30 -bdf2482 -2026-07-30 -bdf2482 -2026-07-30 -ed26197 -2026-07-30 -3ea3059 -2026-07-30 -63becee -2026-07-31 -bf9f729 -2026-07-31 -8aac39f -2026-07-31 -80350a9 -2026-07-31 -4928cbe -2026-07-31 -09277b8 -2026-08-01 -e685be8 -2026-08-02 -10f5608 -2026-08-03 -5771bbc -2026-08-03 -04cfef8 -2026-08-03 -fa8d039 -2026-08-04 -89e05a7 -2026-08-04 -0e073cf -2026-08-04 -09deffa -2026-08-05 -a4b3c97 -2026-08-05 -650dc02 -2026-08-05 -4c59002 -2026-08-05 -fd079e3 -2026-08-05 -e750760 -2026-08-05 -5b7c0c0 -2026-08-05 -b22f23c -2026-08-05 -2d47e73 -2026-08-07 -5db8c6f -2026-08-07 -650a87a -2026-08-08 -cb672c6 -2026-08-09 -6d1db74 -2026-08-09 -6d1db74 -2026-08-09 -dd874e4 -2026-08-09 -149db7d -2026-08-09 -ed59513 -2026-08-09 -12d716e -2026-08-09 -1775fec -2026-08-09 -3663089 -2026-08-10 -0f5117f -2026-08-10 -87442ee -2026-08-10 -ac03a8d -2026-08-10 -2a2efb2 -2026-08-10 -fe88665 -2026-08-11 -912869e -2026-08-11 -1b8fb0a -2026-08-11 -4122b3b -2026-08-11 -06db802 -2026-08-11 -c011d7d -2026-08-11 -9ea6ca2 -2026-08-11 -6a3a7a9 -2026-08-12 -49dbae3 -2026-09-05 -5302ff0 -2026-09-08 -82879ed -2026-09-10 -0577d17 -2026-09-10 -381ba55 -2026-09-11 -386727c -2026-09-11 - -deepseek-v4-flash max stableswap-ng-vyper - -deepseek-v4-flash max venus-isolated-pools-hardhat - -deepseek-v4-flash max very-liquid-vaults-foundry - -deepseek-v4-pro max stableswap-ng-vyper - -deepseek-v4-pro max venus-isolated-pools-hardhat - -deepseek-v4-pro max very-liquid-vaults-foundry - -gpt-5.6-luna high stableswap-ng-vyper - -gpt-5.6-luna high venus-isolated-pools-hardhat - -gpt-5.6-luna high very-liquid-vaults-foundry - -kimi-k3 max stableswap-ng-vyper - -kimi-k3 max venus-isolated-pools-hardhat - -kimi-k3 max very-liquid-vaults-foundry - diff --git a/docs/assets/eval-history/wall-clock-time.svg b/docs/assets/eval-history/wall-clock-time.svg deleted file mode 100644 index 8e568bbe0..000000000 --- a/docs/assets/eval-history/wall-clock-time.svg +++ /dev/null @@ -1,780 +0,0 @@ - - -Wall-clock time (seconds) -Wall-clock time (seconds) by candidate commit and benchmark target; partial values use hollow dashed markers, legacy partial values without a number use a dashed ring and partial n/a label, and unavailable values use an n/a cross - -Wall-clock time (seconds) -ultrafuzz-bench · smoke -Each line tracks one benchmark target across evenly spaced candidate-run columns. - - - -0 - -1293.53 - -2587.06 - -3880.58 - -5174.11 -ultrafuzz-bench smoke cohort-cf3da608 policy-b0f7d6d0 - -cohort-cf3da608 -ultrafuzz-bench smoke cohort-a98bf547 policy-33b8d296 - -cohort-a98bf547 -ultrafuzz-bench smoke cohort-7d061c80 policy-33b8d296 - -cohort-7d061c80 -ultrafuzz-bench smoke cohort-b05ba222 policy-33b8d296 - -cohort-b05ba222 -ultrafuzz-bench smoke cohort-c0e17843 policy-33b8d296 - -cohort-c0e17843 -ultrafuzz-bench smoke cohort-d49aa3e2 policy-33b8d296 - -cohort-d49aa3e2 -ultrafuzz-bench smoke cohort-5d6ac3fd policy-fbbf61e0 - -cohort-5d6ac3fd -ultrafuzz-bench smoke cohort-21f261d4 policy-dfea2bb8 - -cohort-21f261d4 -ultrafuzz-bench smoke cohort-f3e78e3d policy-a7379336 - -cohort-f3e78e3d -deepseek-v4-flash max stableswap-ng-vyper 6d1db74: 1732.26 - - -deepseek-v4-flash max venus-isolated-pools-hardhat 6d1db74: 1036.12 - - -deepseek-v4-flash max very-liquid-vaults-foundry 6d1db74: 1377.72 - - -deepseek-v4-pro max stableswap-ng-vyper 09277b8: unavailable (node-attempt-timestamps-unavailable) - - -n/a 09277b8 -deepseek-v4-pro max venus-isolated-pools-hardhat 09277b8: 1379.93 - - -deepseek-v4-pro max very-liquid-vaults-foundry 09277b8: 1862.11 - - - -gpt-5.6-luna high stableswap-ng-vyper 953a397: 442.15 - - -gpt-5.6-luna high stableswap-ng-vyper 2051b64: 491.97 - - -gpt-5.6-luna high stableswap-ng-vyper 23ef040: 479.21 - - -gpt-5.6-luna high stableswap-ng-vyper 8483637: 467.27 - - -gpt-5.6-luna high stableswap-ng-vyper e03c176: 580.22 - - -gpt-5.6-luna high stableswap-ng-vyper da5d191: 450.12 - - -gpt-5.6-luna high stableswap-ng-vyper 24b90c4: 637.91 - - -gpt-5.6-luna high stableswap-ng-vyper 5cfa918: 491.78 - - -gpt-5.6-luna high stableswap-ng-vyper bdf2482: 580.3 - - -gpt-5.6-luna high stableswap-ng-vyper ed26197: 637.38 - - -gpt-5.6-luna high stableswap-ng-vyper 3ea3059: 715.59 - - -gpt-5.6-luna high stableswap-ng-vyper 63becee: 648.36 - - -gpt-5.6-luna high stableswap-ng-vyper bf9f729: 735.84 - - -gpt-5.6-luna high stableswap-ng-vyper 8aac39f: 497.17 - - -gpt-5.6-luna high stableswap-ng-vyper 80350a9: 723.27 - - -gpt-5.6-luna high stableswap-ng-vyper 4928cbe: 395.92 - - -gpt-5.6-luna high stableswap-ng-vyper e685be8: 503.72 - - -gpt-5.6-luna high stableswap-ng-vyper 10f5608: 620.49 - - -gpt-5.6-luna high stableswap-ng-vyper 04cfef8: 716.91 - - -gpt-5.6-luna high stableswap-ng-vyper fa8d039: 559.42 - - -gpt-5.6-luna high stableswap-ng-vyper 89e05a7: 889.81 - - -gpt-5.6-luna high stableswap-ng-vyper 0e073cf: 480.74 - - -gpt-5.6-luna high stableswap-ng-vyper 09deffa: 433.42 - - -gpt-5.6-luna high stableswap-ng-vyper a4b3c97: 547.36 - - -gpt-5.6-luna high stableswap-ng-vyper 650dc02: 483.73 - - -gpt-5.6-luna high stableswap-ng-vyper 4c59002: 410.77 - - -gpt-5.6-luna high stableswap-ng-vyper fd079e3: 460.94 - - -gpt-5.6-luna high stableswap-ng-vyper e750760: 753.66 - - -gpt-5.6-luna high stableswap-ng-vyper 5b7c0c0: 449.24 - - -gpt-5.6-luna high stableswap-ng-vyper b22f23c: 596.51 - - -gpt-5.6-luna high stableswap-ng-vyper 2d47e73: 520.88 - - -gpt-5.6-luna high stableswap-ng-vyper 5db8c6f: unavailable (node-attempt-timestamps-unavailable) - - -n/a 5db8c6f -gpt-5.6-luna high stableswap-ng-vyper 650a87a: unavailable (node-attempt-timestamps-unavailable) - - -n/a 650a87a -gpt-5.6-luna high stableswap-ng-vyper cb672c6: 486.7 - - -gpt-5.6-luna high stableswap-ng-vyper 6d1db74: 602.41 - - -gpt-5.6-luna high stableswap-ng-vyper dd874e4: 459.82 - - -gpt-5.6-luna high stableswap-ng-vyper 149db7d: 645.39 - - -gpt-5.6-luna high stableswap-ng-vyper ed59513: 493.91 - - -gpt-5.6-luna high stableswap-ng-vyper 12d716e: 803.68 partial (node-attempt-timestamps-final-attempt-only) - - -gpt-5.6-luna high stableswap-ng-vyper 1775fec: 504.16 - - -gpt-5.6-luna high stableswap-ng-vyper 3663089: 508.41 - - -gpt-5.6-luna high stableswap-ng-vyper 0f5117f: 558.13 - - -gpt-5.6-luna high stableswap-ng-vyper 87442ee: 874.38 - - -gpt-5.6-luna high stableswap-ng-vyper ac03a8d: 792.83 - - -gpt-5.6-luna high stableswap-ng-vyper 2a2efb2: 546.36 - - -gpt-5.6-luna high stableswap-ng-vyper fe88665: 667.56 - - -gpt-5.6-luna high stableswap-ng-vyper 912869e: 720.69 - - -gpt-5.6-luna high stableswap-ng-vyper 1b8fb0a: 668.39 - - -gpt-5.6-luna high stableswap-ng-vyper 4122b3b: 605.18 - - -gpt-5.6-luna high stableswap-ng-vyper 06db802: 678.2 - - -gpt-5.6-luna high stableswap-ng-vyper c011d7d: 883.72 - - -gpt-5.6-luna high stableswap-ng-vyper 9ea6ca2: 776.35 - - -gpt-5.6-luna high stableswap-ng-vyper 6a3a7a9: 624.29 - - -gpt-5.6-luna high stableswap-ng-vyper 49dbae3: 1841.75 - - -gpt-5.6-luna high stableswap-ng-vyper 5302ff0: 1784.15 - - -gpt-5.6-luna high stableswap-ng-vyper 82879ed: 1737.97 - - -gpt-5.6-luna high stableswap-ng-vyper 0577d17: 2161.65 - - -gpt-5.6-luna high stableswap-ng-vyper 381ba55: 1711.51 - - -gpt-5.6-luna high stableswap-ng-vyper 386727c: 1962.05 - - - -gpt-5.6-luna high venus-isolated-pools-hardhat 953a397: 508.54 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2051b64: 495.87 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 23ef040: 479.75 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8483637: 428.56 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e03c176: 539.27 - - -gpt-5.6-luna high venus-isolated-pools-hardhat da5d191: 473.22 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 24b90c4: 476.76 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5cfa918: 1097.51 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bdf2482: 583.46 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed26197: 533.54 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3ea3059: 604.77 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 63becee: 817.37 - - -gpt-5.6-luna high venus-isolated-pools-hardhat bf9f729: 667.24 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 8aac39f: 497.62 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 80350a9: 722.71 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4928cbe: 461.65 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e685be8: 436.31 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 10f5608: 440.53 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 04cfef8: unavailable (node-attempt-timestamps-unavailable) - - -n/a 04cfef8 -gpt-5.6-luna high venus-isolated-pools-hardhat fa8d039: 740.7 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 89e05a7: 531.12 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0e073cf: 481.76 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 09deffa: 414.21 - - -gpt-5.6-luna high venus-isolated-pools-hardhat a4b3c97: 904.94 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650dc02: 521.36 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4c59002: 490.58 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fd079e3: 335.95 - - -gpt-5.6-luna high venus-isolated-pools-hardhat e750760: 791.87 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5b7c0c0: 560.2 - - -gpt-5.6-luna high venus-isolated-pools-hardhat b22f23c: 650.87 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2d47e73: 416.63 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5db8c6f: 385.09 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 650a87a: unavailable (node-attempt-timestamps-unavailable) - - -n/a 650a87a -gpt-5.6-luna high venus-isolated-pools-hardhat cb672c6: 453.7 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6d1db74: 508.84 partial (node-attempt-timestamps-final-attempt-only) - - -gpt-5.6-luna high venus-isolated-pools-hardhat dd874e4: 400.03 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 149db7d: 517.49 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ed59513: 582.78 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 12d716e: 676.66 partial (node-attempt-timestamps-final-attempt-only) - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1775fec: 533.66 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 3663089: 524.22 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0f5117f: 461.34 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 87442ee: 939.66 - - -gpt-5.6-luna high venus-isolated-pools-hardhat ac03a8d: 867.95 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 2a2efb2: 577.24 - - -gpt-5.6-luna high venus-isolated-pools-hardhat fe88665: 653.31 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 912869e: 750.85 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 1b8fb0a: 727.07 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 4122b3b: 609.42 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 06db802: 789.68 - - -gpt-5.6-luna high venus-isolated-pools-hardhat c011d7d: 906.25 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 9ea6ca2: 885.61 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 6a3a7a9: 663.02 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 49dbae3: 2005.97 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 5302ff0: 2032.37 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 82879ed: 1971.97 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 0577d17: 2015.11 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 381ba55: 1937.97 - - -gpt-5.6-luna high venus-isolated-pools-hardhat 386727c: 2023.7 - - - -gpt-5.6-luna high very-liquid-vaults-foundry 953a397: 571.4 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2051b64: 561.22 - - -gpt-5.6-luna high very-liquid-vaults-foundry 23ef040: 583.98 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8483637: 515.56 - - -gpt-5.6-luna high very-liquid-vaults-foundry e03c176: 670.92 - - -gpt-5.6-luna high very-liquid-vaults-foundry da5d191: 581.5 - - -gpt-5.6-luna high very-liquid-vaults-foundry 24b90c4: 505.55 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5cfa918: 584.97 - - -gpt-5.6-luna high very-liquid-vaults-foundry bdf2482: 583.61 - - -gpt-5.6-luna high very-liquid-vaults-foundry ed26197: 640.01 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3ea3059: 649.53 - - -gpt-5.6-luna high very-liquid-vaults-foundry 63becee: 817.08 - - -gpt-5.6-luna high very-liquid-vaults-foundry bf9f729: 735.55 - - -gpt-5.6-luna high very-liquid-vaults-foundry 8aac39f: 673.18 - - -gpt-5.6-luna high very-liquid-vaults-foundry 80350a9: 702.34 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4928cbe: 569 - - -gpt-5.6-luna high very-liquid-vaults-foundry e685be8: 593.5 - - -gpt-5.6-luna high very-liquid-vaults-foundry 10f5608: 575.92 - - -gpt-5.6-luna high very-liquid-vaults-foundry 04cfef8: 808.4 - - -gpt-5.6-luna high very-liquid-vaults-foundry fa8d039: 830.75 - - -gpt-5.6-luna high very-liquid-vaults-foundry 89e05a7: 708.38 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0e073cf: unavailable (node-attempt-timestamps-unavailable) - - -n/a 0e073cf -gpt-5.6-luna high very-liquid-vaults-foundry 09deffa: 522.16 - - -gpt-5.6-luna high very-liquid-vaults-foundry a4b3c97: 475.49 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650dc02: 538.61 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4c59002: 626.88 - - -gpt-5.6-luna high very-liquid-vaults-foundry fd079e3: 568.71 - - -gpt-5.6-luna high very-liquid-vaults-foundry e750760: 568.64 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5b7c0c0: unavailable (node-attempt-timestamps-unavailable) - - -n/a 5b7c0c0 -gpt-5.6-luna high very-liquid-vaults-foundry b22f23c: unavailable (node-attempt-timestamps-unavailable) - - -n/a b22f23c -gpt-5.6-luna high very-liquid-vaults-foundry 2d47e73: unavailable (node-attempt-timestamps-unavailable) - - -n/a 2d47e73 -gpt-5.6-luna high very-liquid-vaults-foundry 5db8c6f: 419.93 - - -gpt-5.6-luna high very-liquid-vaults-foundry 650a87a: 433.42 - - -gpt-5.6-luna high very-liquid-vaults-foundry cb672c6: 515.16 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6d1db74: 530.45 - - -gpt-5.6-luna high very-liquid-vaults-foundry dd874e4: 575.95 partial (node-attempt-timestamps-final-attempt-only) - - -gpt-5.6-luna high very-liquid-vaults-foundry 149db7d: 660.96 partial (node-attempt-timestamps-final-attempt-only) - - -gpt-5.6-luna high very-liquid-vaults-foundry ed59513: 661.94 partial (node-attempt-timestamps-final-attempt-only) - - -gpt-5.6-luna high very-liquid-vaults-foundry 12d716e: 741.36 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1775fec: 681.23 - - -gpt-5.6-luna high very-liquid-vaults-foundry 3663089: 561.42 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0f5117f: 713.31 - - -gpt-5.6-luna high very-liquid-vaults-foundry 87442ee: 987.17 - - -gpt-5.6-luna high very-liquid-vaults-foundry ac03a8d: 925.51 - - -gpt-5.6-luna high very-liquid-vaults-foundry 2a2efb2: 792.86 - - -gpt-5.6-luna high very-liquid-vaults-foundry fe88665: 748.03 - - -gpt-5.6-luna high very-liquid-vaults-foundry 912869e: 850.46 - - -gpt-5.6-luna high very-liquid-vaults-foundry 1b8fb0a: 1059.98 - - -gpt-5.6-luna high very-liquid-vaults-foundry 4122b3b: 803.99 - - -gpt-5.6-luna high very-liquid-vaults-foundry 06db802: 841.21 - - -gpt-5.6-luna high very-liquid-vaults-foundry c011d7d: 1356.86 partial (node-attempt-timestamps-final-attempt-only) - - -gpt-5.6-luna high very-liquid-vaults-foundry 9ea6ca2: 708.44 - - -gpt-5.6-luna high very-liquid-vaults-foundry 6a3a7a9: 806.07 - - -gpt-5.6-luna high very-liquid-vaults-foundry 49dbae3: 1944.58 - - -gpt-5.6-luna high very-liquid-vaults-foundry 5302ff0: 1878.01 - - -gpt-5.6-luna high very-liquid-vaults-foundry 82879ed: 1968.76 - - -gpt-5.6-luna high very-liquid-vaults-foundry 0577d17: 2344.58 - - -gpt-5.6-luna high very-liquid-vaults-foundry 381ba55: 1944.12 - - -gpt-5.6-luna high very-liquid-vaults-foundry 386727c: 1657.2 - - - -kimi-k3 max stableswap-ng-vyper bdf2482: 4966.17 - - -kimi-k3 max stableswap-ng-vyper 5771bbc: 3288.13 - - - -kimi-k3 max venus-isolated-pools-hardhat bdf2482: 4746.34 - - -kimi-k3 max venus-isolated-pools-hardhat 5771bbc: 2803.37 - - - -kimi-k3 max very-liquid-vaults-foundry bdf2482: 5174.11 - - -kimi-k3 max very-liquid-vaults-foundry 5771bbc: 2319.22 - - -953a397 -2026-07-21 -2051b64 -2026-07-22 -23ef040 -2026-07-22 -8483637 -2026-07-22 -e03c176 -2026-07-23 -da5d191 -2026-07-23 -24b90c4 -2026-07-30 -5cfa918 -2026-07-30 -bdf2482 -2026-07-30 -bdf2482 -2026-07-30 -ed26197 -2026-07-30 -3ea3059 -2026-07-30 -63becee -2026-07-31 -bf9f729 -2026-07-31 -8aac39f -2026-07-31 -80350a9 -2026-07-31 -4928cbe -2026-07-31 -09277b8 -2026-08-01 -e685be8 -2026-08-02 -10f5608 -2026-08-03 -5771bbc -2026-08-03 -04cfef8 -2026-08-03 -fa8d039 -2026-08-04 -89e05a7 -2026-08-04 -0e073cf -2026-08-04 -09deffa -2026-08-05 -a4b3c97 -2026-08-05 -650dc02 -2026-08-05 -4c59002 -2026-08-05 -fd079e3 -2026-08-05 -e750760 -2026-08-05 -5b7c0c0 -2026-08-05 -b22f23c -2026-08-05 -2d47e73 -2026-08-07 -5db8c6f -2026-08-07 -650a87a -2026-08-08 -cb672c6 -2026-08-09 -6d1db74 -2026-08-09 -6d1db74 -2026-08-09 -dd874e4 -2026-08-09 -149db7d -2026-08-09 -ed59513 -2026-08-09 -12d716e -2026-08-09 -1775fec -2026-08-09 -3663089 -2026-08-10 -0f5117f -2026-08-10 -87442ee -2026-08-10 -ac03a8d -2026-08-10 -2a2efb2 -2026-08-10 -fe88665 -2026-08-11 -912869e -2026-08-11 -1b8fb0a -2026-08-11 -4122b3b -2026-08-11 -06db802 -2026-08-11 -c011d7d -2026-08-11 -9ea6ca2 -2026-08-11 -6a3a7a9 -2026-08-12 -49dbae3 -2026-09-05 -5302ff0 -2026-09-08 -82879ed -2026-09-10 -0577d17 -2026-09-10 -381ba55 -2026-09-11 -386727c -2026-09-11 - -deepseek-v4-flash max stableswap-ng-vyper - -deepseek-v4-flash max venus-isolated-pools-hardhat - -deepseek-v4-flash max very-liquid-vaults-foundry - -deepseek-v4-pro max stableswap-ng-vyper - -deepseek-v4-pro max venus-isolated-pools-hardhat - -deepseek-v4-pro max very-liquid-vaults-foundry - -gpt-5.6-luna high stableswap-ng-vyper - -gpt-5.6-luna high venus-isolated-pools-hardhat - -gpt-5.6-luna high very-liquid-vaults-foundry - -kimi-k3 max stableswap-ng-vyper - -kimi-k3 max venus-isolated-pools-hardhat - -kimi-k3 max very-liquid-vaults-foundry - diff --git a/docs/config.md b/docs/config.md index 4ccc55e3e..1b3842b75 100644 --- a/docs/config.md +++ b/docs/config.md @@ -50,11 +50,17 @@ project primary count. The shipped `default` profile uses three attempts; five, and `invariant-only` inherits three. An explicit project `[retry]` value overrides the profile. The complete primary-plus-fallback chain may contain at most 100 attempts. Omitting `agents`, or leaving it empty, keeps model fallback disabled. -Retries use bounded exponential backoff, a fresh session, and the same effective -task prompt, including Smithers' safety contracts; Ultrafuzz does not inspect -provider error text. The -planned chain and actual producer are recorded in the task manifest, attempt -ledger, and final report. +A retry waits one minute, then two, then four, and at most five minutes (the +Smithers cap), and uses a fresh session and the same effective task prompt, +including Smithers' safety contracts; Ultrafuzz does not inspect provider error +text. Repeated identical failures do not end the planned chain early; Smithers +still stops it at a failure it classifies as non-retryable, such as a CLI +configuration or authentication error, and pauses the run on a provider quota +limit. Dependency admission is not retried: it re-reads the same producer files, +so an admission failure, including a file-system or validator error while +reading them, fails the task without a retry or fallback. The planned chain and +actual producer are recorded in the task manifest, attempt ledger, and final +report. Retry chains currently require local execution. Cloud planning accepts one effective attempt, and local fallback across different agent implementations @@ -65,7 +71,7 @@ Agent configuration and model profiles may name only `ClaudeAgent`, `CodexAgent` `DeepSeekAgent`, `KimiAgent`, `OpenCodeAgent`, `OpenRouterAgent`, or `PiAgent`. The complete `.smithers/agents` tree must byte-match the packaged stock closure; custom adapters and registries -are unsupported, and `ultrafuzz init --force` restores the authenticated copy. The stock closure always uses YOLO/bypass-permissions; stricter per-project adapters are unsupported. +are unsupported, and `ultrafuzz init` restores the authenticated copy. The stock closure always uses YOLO/bypass-permissions; stricter per-project adapters are unsupported. Each stock agent's `api_key_env` must use its canonical provider credential name. Custom environment variable names fail config validation. @@ -198,7 +204,9 @@ Kimi's four components — uncached input, output, cache reads, and cache creation — are reported independently; Kimi already folds thinking tokens into output, so no separate reasoning total is published. Malformed or absent usage stays absent rather than becoming zeros, which keeps accounting honest about -what it does not know. Kimi model pricing resolves against the Moonshot +what it does not know. An unreadable wire, including inherited history torn by +a killed attempt, leaves that invocation's usage absent instead of failing the +invocation. Kimi model pricing resolves against the Moonshot provider entry in the pricing catalog, so the configured alias must match a Moonshot catalog model id such as `kimi-k3`; anything else is reported as an unresolved model instead of being priced from a same-named third-party entry. @@ -247,9 +255,10 @@ value is rejected before execution. See DeepSeek's and [Anthropic API guide](https://api-docs.deepseek.com/guides/anthropic_api). DeepSeek's automatic disk cache reports cache misses and hits independently. -Ultrafuzz records those as uncached input and cache-read tokens, records no -cache-write charge, and treats the provider's output count as already including -thinking tokens rather than publishing a second reasoning component. Pricing +Claude Code reports them under Anthropic field names (`input_tokens`, +`cache_read_input_tokens`), which the pinned Smithers Claude Code adapter +already reads, so the DeepSeek adapter does no usage parsing of its own. The +output count already includes thinking tokens. Pricing is pinned to the first-party `deepseek` catalog entry so a same-named hosted or subscription plan cannot supply a zero or unrelated rate. The current [DeepSeek price table](https://api-docs.deepseek.com/quick_start/pricing) lists @@ -260,7 +269,8 @@ million cache-hit input tokens, and $0.87 per million output tokens. `ultrafuzz init` also generates a dedicated `OpenCodeAgent`. It is opt-in and non-default: nothing selects it until a topology group or `--agent` names it. -The default root config includes an opt-in OpenCode profile: +The default root config has the `[agents.OpenCodeAgent]` block below but no +OpenCode model profile, so add one such as `[models.opencode]` to use it: ```toml [models.opencode] @@ -321,10 +331,9 @@ else. `config_dir` is the XDG parent, not OpenCode's own directory: `XDG_DATA_HOME` becomes `/data`, so pointing it at `~/.config/opencode` or `~/.local/share/opencode` picks nothing up. -`ultrafuzz doctor` requires the `opencode` executable whenever any configured -profile uses `OpenCodeAgent` — every profile in `[models.*]` is checked, not -only the one a run selects, so keeping the shipped `[models.opencode]` profile -means every contributor needs the CLI installed. +`ultrafuzz doctor` requires the `opencode` executable only when a node of the +selected topology, or its `[retry] agents` fallback chain, uses an +`OpenCodeAgent` profile; otherwise the executable is listed as not required. Default triage requires quorum `3` from a panel size of `4`: @@ -376,6 +385,50 @@ execution. The Smithers type surface pinned by this release stops at `xhigh`, but pi's command surface also accepts `max`, so the adapter validates and forwards that final level without degrading it. +The adapter counts usage from each assistant `message_end` event. A response +whose usage Pi reports inconsistently (token counts that do not add up to +`totalTokens`, or a cost breakdown that is missing or does not add up), or +whose counts are invalid or would overflow the running totals, is left out +whole instead of failing the invocation, so that invocation's usage is then a +lower bound. + +## OpenRouter guardrails + +`OpenRouterAgent`, `PiAgent`, and `OpenCodeAgent` with an `openrouter/` model +send their requests through OpenRouter with the key in `OPENROUTER_API_KEY`. If +any guardrail covering that key sets +[prompt-injection detection](https://openrouter.ai/docs/guides/features/guardrails/prompt-injection) +to **Block**, OpenRouter rejects each request its detector matches with HTTP +403 `Request blocked: prompt injection patterns detected` before it reaches a +model. + +The match need not be in Ultrafuzz's task prompt. These harnesses also send +their own system prompts and the target source and test output the agent reads, +and by default OpenRouter scans every message in a request, including base64- +and hex-decoded text. For example, the default system prompt of OpenCode +1.18.18, used for models without a model-specific prompt such as DeepSeek, +Qwen, or GLM, has an `assistant: [...]` line followed by a `user:` line, which +matches OpenRouter's documented `role_delimiter_injection` pattern. + +In the **Security** section of every guardrail that covers the key (the +workspace default and any member or API-key guardrail), set prompt-injection +detection to **Flag**, which records matches without enforcing them, or turn it +off. OpenRouter applies the most restrictive action when several guardrails +apply. Do not use **Redact** either: it replaces each match with +`[PROMPT_INJECTION]` and forwards the request, so the model can work from +altered source or tool output with no error for Ultrafuzz to report. + +The workspace default covers every key in its workspace and a member guardrail +every key of that member, so relaxing either can affect more than Ultrafuzz. +Creating the Ultrafuzz key in a workspace of its own confines the +workspace-default change to that key. In an organization account, only an +organization admin can change guardrails. + +Ultrafuzz has no special handling for this rejection: the attempt fails like any +other agent error and follows the `[retry]` policy above. A retry on the same +profile sends the same task prompt with the same key, so it is rejected again +when the match is in that prompt or in the harness's system prompt. + ## Forge process guard Worker environments put a run-scoped Forge wrapper ahead of the installed @@ -458,10 +511,8 @@ trusted local execution model remains unchanged. ## Redaction Run artifacts store redacted resolved config and a redaction manifest. -Sensitive model values are redacted before persistence. Launch guards for -literal redaction placeholders may fail before workflow launch when enabled, -and manifest entries mark values that must be restored from current config -before launch. +Sensitive model values are redacted before persistence. The manifest records +which values were redacted; no command restores values from it. ## Eval suites diff --git a/docs/contributing.md b/docs/contributing.md index 7251c611a..04f9cedd0 100644 --- a/docs/contributing.md +++ b/docs/contributing.md @@ -78,3 +78,12 @@ pnpm -r test Prefer narrow package checks while iterating, then run broader gates before PR handoff when a change touches shared behavior or release workflows. + +## Complexity Ceiling + +`pnpm -w lint` fails any function under `packages/` or `scripts/` whose +cyclomatic complexity is above 90, the highest value in the codebase when the +ceiling was added. There are no suppressions: lower the ceiling in +`eslint.config.js` when the most complex functions are simplified. +`pnpm -w lint:strict:ci` holds changed lines to the stricter complexity, size, +and type-aware budgets. diff --git a/docs/explanation/campaigns.md b/docs/explanation/campaigns.md index 07b98c9fc..7e718b210 100644 --- a/docs/explanation/campaigns.md +++ b/docs/explanation/campaigns.md @@ -97,6 +97,5 @@ automatically counted as false positives. Ultrafuzz ships this methodology as a product surface: eval suites run a target × variant × trial matrix against ground-truth bugs, score precision, -recall, and F1 locally, and optionally mirror node telemetry to an eval cloud -provider. See [Eval Suites](../reference/evals.md) and +recall, and F1 locally. See [Eval Suites](../reference/evals.md) and [Run Eval Suites](../how-to/run-evals.md). diff --git a/docs/explanation/topology-prompts-artifacts.md b/docs/explanation/topology-prompts-artifacts.md index 5c7ad740f..0f3e3b24a 100644 --- a/docs/explanation/topology-prompts-artifacts.md +++ b/docs/explanation/topology-prompts-artifacts.md @@ -67,8 +67,8 @@ This makes handoffs explicit: - The prompt tells the agent what to write. - The topology declares a named, versioned contract for the file. -- Planning binds that contract to an exact schema ID, schema digest, bundle - digest, and validator build. +- Planning binds that contract to an exact schema ID, schema digest, and bundle + digest, and records the validator build that planned it. - Every JSON producer runs the displayed schema-validation command after its final write and before returning. - A deterministic workflow task validates the artifact before dependents start. @@ -77,8 +77,8 @@ This makes handoffs explicit: - Downstream prompts reference it through typed template helpers. The displayed command resolves through a host-managed launcher placed before -target-controlled `PATH` entries. A real fixture preflight checks that launcher, -the schema registry, and the validator build before model work. The command is +target-controlled `PATH` entries. A real fixture preflight checks that launcher +and the schema identity it validates with before model work. The command is producer feedback, not a new inter-node message or validation-receipt schema. The runtime repeats shape validation and then applies contextual gates. diff --git a/docs/how-to/edit-prompts-topology.md b/docs/how-to/edit-prompts-topology.md index 99ca17a04..d2c3fc106 100644 --- a/docs/how-to/edit-prompts-topology.md +++ b/docs/how-to/edit-prompts-topology.md @@ -14,6 +14,13 @@ After `ultrafuzz init`, editable prompts live under: Use the [prompt catalog](../reference/prompt-catalog.md) to find every shipped workflow prompt and its role before choosing what to customize. +Runs use these project copies, and `ultrafuzz init` keeps them unless you pass +`--force`, which also overwrites `ultrafuzz.toml` and the topology. After an +upgrade a copy therefore keeps the text of the release that scaffolded it. +`ultrafuzz validate` warns about every project prompt that differs from the +built-in prompt at the same path. To take the built-in version of a prompt, +delete your copy and rerun `ultrafuzz init`. + Prompt frontmatter may include only identity and display metadata: ```md @@ -84,6 +91,13 @@ nodes: Agentic nodes run through the configured workflow adapter. Topology does not define arbitrary shell runners. +`ultrafuzz init` also keeps an existing `.ultrafuzz/topology.yml`, so after an +upgrade it still has the topology of the release that scaffolded it. If you +have not customized it, +`ultrafuzz topology copy default .ultrafuzz/topology.yml --force` replaces it +with the current packaged default and leaves `ultrafuzz.toml` and the prompts +alone; otherwise, merge the release's topology changes by hand. + ## Declare Durable Handoffs Use `outputs` for files a node must write under its artifact directory. Every diff --git a/docs/how-to/restart-continue.md b/docs/how-to/restart-continue.md index c539fd66f..0e8c5b2ed 100644 --- a/docs/how-to/restart-continue.md +++ b/docs/how-to/restart-continue.md @@ -43,6 +43,7 @@ state: ultrafuzz resume --project /path/to/target-protocol ultrafuzz resume --project /path/to/target-protocol --max-concurrency 4 ultrafuzz resume --project /path/to/target-protocol --refresh-controller +ultrafuzz resume --project /path/to/target-protocol --retry-failed ultrafuzz resume --project /path/to/target-protocol \ --reset-node node:failed-task --max-concurrency 4 ``` @@ -72,8 +73,14 @@ Completed node attempts remain in `attempts.jsonl` across every continuation. append-only ledger rather than from a mutable lifecycle counter. Use `--reset-node` to explicitly retry one failed workflow node and reset its -dependents before the linked run continues. Ordinary resume performs no reset, -timetravel, replay, or fork. +dependents before the linked run continues. `--retry-failed` resets every failed +or stalled node, retrying a failed artifact verifier from its agent producer. +Ordinary resume performs no reset, timetravel, replay, or fork. + +A task skipped because a prerequisite failed is not reused like a finished row: +the resumed workflow decides the skip again from the restored task states. It +stays skipped while the prerequisite is still failed, and runs once a reset lets +the prerequisite succeed. ## Replay A Linked Run diff --git a/docs/how-to/run-evals-on-modal.md b/docs/how-to/run-evals-on-modal.md index 32e28f7a2..89165747f 100644 --- a/docs/how-to/run-evals-on-modal.md +++ b/docs/how-to/run-evals-on-modal.md @@ -1,8 +1,8 @@ # Run evals on Modal Ultrafuzz can run long evaluation rows in Modal sandboxes while retaining -telemetry, scores, and reports in the run workspace. The runner uses the Modal -TypeScript SDK; Python is not required. +scores and reports in the run workspace. The runner uses the Modal TypeScript +SDK; Python is not required. ## Security and storage boundaries diff --git a/docs/how-to/run-evals.md b/docs/how-to/run-evals.md index f7df8ec9c..ffd66f27a 100644 --- a/docs/how-to/run-evals.md +++ b/docs/how-to/run-evals.md @@ -3,7 +3,7 @@ Eval suites benchmark the Ultrafuzz pipeline against targets with known ground-truth bugs and rank prompt or topology variants by precision, recall, and F1. The full loop is `plan → run → score → report → compare`. Evaluation -records, telemetry, scores, and reports stay in local artifacts. +records, scores, and reports stay in local artifacts. ## Configure The Suite @@ -11,7 +11,7 @@ Two files split the configuration: - The eval YAML (default `.ultrafuzz/evals/bug-finding.yml`) is the committable experiment definition: model profiles, targets, variants, trial - counts, grading metrics, and the telemetry policy. It never names a + counts, grading metrics, and whether `run` watches rows. It never names a provider, an endpoint, or an env var. - The `ultrafuzz.toml` `[eval]` section is per-environment: the default suite path and the machine-specific `ground_truth_root`. `provider = "none"` is @@ -67,9 +67,9 @@ ultrafuzz eval run --target-root /path/to/target-checkouts ultrafuzz eval run --row --no-watch ``` -`run` launches an Ultrafuzz run per matrix row (or a `--row` selection), -polls the runs to a terminal state, and records local telemetry when the suite -enables it. `--no-watch` launches detached without polling. Artifacts accumulate under +`run` launches an Ultrafuzz run per matrix row (or a `--row` selection) and, +unless the suite sets `reporting.node_telemetry: false`, polls the runs to a +terminal state. `--no-watch` launches detached without polling. Artifacts accumulate under `.ultrafuzz/evals/runs//`. ## Score And Compare @@ -157,8 +157,8 @@ variant and use the separate GPT-5.6 Sol `xhigh` judge. The benchmark adapter converts either lane into the normal `EvalSuiteSpec` and can project one runner for an isolated Modal pair while retaining the fixed judge. -After a generation finishes and has been scored, append it and regenerate all -nine SVG charts in one transaction: +After a generation finishes and has been scored, append it and regenerate the +three README SVG charts in one transaction: ```bash ultrafuzz eval history \ diff --git a/docs/reference/agent-adapter-boundaries.md b/docs/reference/agent-adapter-boundaries.md index 179ae3fcf..f46bbb2ad 100644 --- a/docs/reference/agent-adapter-boundaries.md +++ b/docs/reference/agent-adapter-boundaries.md @@ -16,22 +16,20 @@ and the gate requires an explicit policy for every adapter actually registered in `agentFactories`, including a factory imported under an alias or registered without a matching re-export. -| Adapter | Baseline | Classification | Existing option or missing surface | Upstream dependency | -| ---------------- | -------: | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------- | -| `claude.tsx` | 86 | Thin mapping. Its override applies Ultrafuzz's child-environment policy but does not rebuild an orchestrator responsibility. | Uses `model`, `extraArgs`, `addDir`, `permissionMode`, `settingSources`, `apiKey`, `configDir`, and `env`. | None. | -| `codex.tsx` | 162 | Partly avoidable. The local argv rewrite and resume awareness exist because a working constructor option is serialized incorrectly upstream. Provider-home inspection and child-environment filtering are local policy. | `addDir` exists, but multiple values become one `--add-dir` occurrence. | [smithers#1622](https://github.com/smithersai/smithers/issues/1622) | -| `deepseek.tsx` | 341 | Inherent after removing its avoidable reasoning-effort argv mapping. Route, auth, and effort are now thin mappings; result parsing and token normalization have no typed upstream surface. | Uses `model`, first-class `effort`, `addDir`, `permissionMode`, `settingSources`, `env`, and `configDir`; a custom-provider usage normalizer is missing. | [smithers#1624](https://github.com/smithersai/smithers/issues/1624) | -| `kimi.tsx` | 1,520 | Inherent with the current dependency except for Ultrafuzz-specific bounded-I/O and credential-governance checks. Usage discovery, actual-session recovery, argv compatibility, and runtime-home isolation cannot be expressed by constructor options. | `model`, `extraArgs`, `env`, `configDir`, and `session` exist; invocation-local usage, actual-session resolution, separate credential/runtime homes, and a Kimi Code 0.29.x command dialect are missing. | [smithers#1623](https://github.com/smithersai/smithers/issues/1623), [smithers#1626](https://github.com/smithersai/smithers/issues/1626) | -| `openrouter.tsx` | 1,234 | Inherent with the current dependency except for local credential/config materialization. Provider-output quarantine and exact-session retry are orchestration responsibilities; the inherited Codex argv workaround is separately avoidable. | `config`, `configDir`, `env`, `model`, and `addDir` cover the route; a bounded provider-recovery policy is missing. | [smithers#1622](https://github.com/smithersai/smithers/issues/1622), [smithers#1625](https://github.com/smithersai/smithers/issues/1625) | +| Adapter | Classification | Existing option or missing surface | Upstream dependency | +| ---------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------- | +| `claude.tsx` | Thin mapping. Its override applies Ultrafuzz's child-environment policy but does not rebuild an orchestrator responsibility. | Uses `model`, `extraArgs`, `addDir`, `permissionMode`, `settingSources`, `apiKey`, `configDir`, and `env`. | None. | +| `codex.tsx` | Partly avoidable. The local argv rewrite and resume awareness exist because a working constructor option is serialized incorrectly upstream. Provider-home inspection and child-environment filtering are local policy. | `addDir` exists, but multiple values become one `--add-dir` occurrence. | [smithers#1622](https://github.com/smithersai/smithers/issues/1622) | +| `deepseek.tsx` | Thin mapping. Route, auth, and effort map onto constructor options; Claude Code reports usage under Anthropic field names that Smithers already reads, so the adapter parses no output. | Uses `model`, first-class `effort`, `addDir`, `permissionMode`, `settingSources`, `env`, and `configDir`. | None. | +| `kimi.tsx` | Inherent with the current dependency except for Ultrafuzz-specific bounded-I/O and credential-governance checks. Usage discovery, actual-session recovery, argv compatibility, and runtime-home isolation cannot be expressed by constructor options. | `model`, `extraArgs`, `env`, `configDir`, and `session` exist; invocation-local usage, actual-session resolution, separate credential/runtime homes, and a Kimi Code 0.29.x command dialect are missing. | [smithers#1623](https://github.com/smithersai/smithers/issues/1623), [smithers#1626](https://github.com/smithersai/smithers/issues/1626) | +| `openrouter.tsx` | Inherent with the current dependency except for local credential/config materialization. Provider-output quarantine and exact-session retry are orchestration responsibilities; the inherited Codex argv workaround is separately avoidable. | `config`, `configDir`, `env`, `model`, and `addDir` cover the route; a bounded provider-recovery policy is missing. | [smithers#1622](https://github.com/smithersai/smithers/issues/1622), [smithers#1625](https://github.com/smithersai/smithers/issues/1625) | -The line ceilings deliberately allow only a small formatting margin: Claude -100, Codex 175, DeepSeek 350, Kimi 1,525, and OpenRouter 1,250 lines. The listed -responsibilities are explicit review declarations, not conclusions inferred -from comments or prose. The gate supplements them with a conservative AST lower -bound over executable source: static or dynamic filesystem-walking imports, -command `args` access and CLI-looking argument arrays, output-interpreter hooks -and JSON parsing of line-oriented output, common session-continuation fields, -and common token-usage fields. Every detected category must be declared; +The listed responsibilities are explicit review declarations, not conclusions +inferred from comments or prose. The gate supplements them with a conservative +AST lower bound over executable source: static or dynamic filesystem-walking +imports, command `args` access and CLI-looking argument arrays, +output-interpreter hooks and JSON parsing of line-oriented output, common +session-continuation fields, and common token-usage fields. Every detected category must be declared; declarations may include additional inherited or semantically reviewed responsibilities that the detector cannot infer. Type-only declarations, comments, strings outside argument arrays, and simple values or literal option @@ -39,29 +37,23 @@ arrays passed by a `create*Agent` factory into its returned `*Agent` constructor, named config-file reads, and empty diagnostic argv are deliberately excluded. -Every `.ts` and `.tsx` source under the adapter tree has two independent review -records. Its structural policy records a reviewed purpose, TypeScript -syntax-node ceiling, and exact SHA-256 source fingerprint. A separate central -responsibility policy repeats the exact fingerprint beside the declared -responsibilities and upstream links. Any byte change therefore invalidates both -records: updating the ordinary source fingerprint and ceiling cannot reuse a -stale responsibility review, even when the AST lower bound does not recognize -the new semantic form. Refreshing the second literal attests that the central -classification was reviewed; leaving the responsibility set unchanged is an -explicit reviewed-unchanged decision. The syntax count uses the -workspace-pinned TypeScript parser. This is an auditable two-stage source -freeze, not a claim that CI can infer every semantic behavior. +Every `.ts` and `.tsx` source under the adapter tree has one policy record: its +reviewed purpose, its declared responsibilities, and the upstream links that +justify them. The gate does not pin source bytes, line counts, or syntax-node +counts, so a comment edit or a one-line fix needs no policy change; a change +that adds a detected responsibility still fails until the policy declares it. +This is a conservative static lower bound, not a claim that CI can infer every +semantic behavior. The inventory walk is recursive. A new helper, either `.ts` or `.tsx`, fails -until it receives both an explicit structural policy and an independent -responsibility review. Non-adapter helpers must attest an empty responsibility -set and may not contain any of the lower-bound orchestration signals, so moving -such code out of a registered adapter cannot bypass either review layer. A -future shared inherent workaround must add explicit adapter ownership rather -than weakening that default. Any policy update must classify a changed adapter -responsibility in the same reviewed pull request. +until it receives an explicit policy. Non-adapter helpers must attest an empty +responsibility set and may not contain any of the lower-bound orchestration +signals, so moving such code out of a registered adapter cannot bypass the +review. A future shared inherent workaround must add explicit adapter ownership +rather than weakening that default. Any policy update must classify a changed +adapter responsibility in the same reviewed pull request. The gate is `packages/runtime/test/agent-adapter-boundaries.test.ts`. It scans every TypeScript source file in the adapter directory, derives shipped adapters -from `agentFactories`, and runs in both the required pull-request runtime smoke -path and the full runtime supporting-test shard. +from `agentFactories`, and runs in the runtime supporting-test lane, which pull +requests require. diff --git a/docs/reference/artifact-contract-migration-v2.md b/docs/reference/artifact-contract-migration-v2.md index b7e4526a0..8e2e32cda 100644 --- a/docs/reference/artifact-contract-migration-v2.md +++ b/docs/reference/artifact-contract-migration-v2.md @@ -124,9 +124,10 @@ not renamed historical payloads. | EVMBench profile | `ultrafuzz.evmbench.profile.v1` | `ultrafuzz.evmbench.profile.v2`. | | EVMBench result | `ultrafuzz.evmbench.result.v1` | `ultrafuzz.evmbench.result.v2`, returned through the shared CLI v2 envelope. | -Retained evaluation documents for ground truth, publication/status, -recovery-equivalence, telemetry cursors, automatic history publication, and -benchmark provenance now have registered closed schemas. New typed analysis +Retained evaluation documents for ground truth, status, recovery-equivalence, +and benchmark provenance now have registered closed schemas; the +publication-state, telemetry-cursor, and automatic history-publication schemas +were later deleted as unused. New typed analysis documents include adjudication handoff, finding manifest, instance clusters, ground-truth credits, benchmark provenance, benchmark source/analysis manifests, and the embedded verified-report authority used by scoring. Their v1 diff --git a/docs/reference/artifacts-reports.md b/docs/reference/artifacts-reports.md index 2683bb70e..656258f40 100644 --- a/docs/reference/artifacts-reports.md +++ b/docs/reference/artifacts-reports.md @@ -30,14 +30,18 @@ trusted-cli.json trusted-bin/ artifacts/ review/ -events.index/ workspaces/ workspaces.json ``` -`source-run.json` is present when the run derives from another run. Event query -indexes are JSONL files derived from `events.jsonl`; SQLite events are not part -of the artifact contract. +`source-run.json` is present when the run derives from another run. +`events.jsonl` is the only event journal: event queries filter it, and SQLite +events are not part of the artifact contract. An append checks the new event +against the final event and any trailing events with the same timestamp; +`replayEvents` and `queryEvents` validate the whole journal. The journal has no +record-count limit; its 64 MiB byte limit still applies. Runs created before +this change may also have an `events.index/` directory. Nothing reads it, and +report bundles still copy it. `usage.jsonl` is an append-only ledger of normalized workflow usage events. Each entry's immutable identity is the exact Smithers pair @@ -173,6 +177,12 @@ Node statuses are: - `reused-from-prior-run` - `invalidated` +Synchronization records a node as `timed-out` from Smithers' typed deadline +codes (`TASK_TIMEOUT`, `TASK_HEARTBEAT_TIMEOUT`, `PROCESS_TIMEOUT`, +`PROCESS_IDLE_TIMEOUT`) and heartbeat-timeout events, not from error text: a +failure whose message mentions a timeout, or a deadline reported only as text +such as a Modal cloud-node deadline, is `failed`. + Every nonterminal node records `wait_since`, a typed `wait_reason`, and a typed `next_eligible_action`. Wait reasons distinguish ready work, capacity and dependency waits, retry backoff, external gates, controller loss, and active @@ -204,17 +214,44 @@ manifests. Each new attempt also records its selected Smithers chain index, model-profile ID, agent reference, optional model/reasoning values, and -primary-or-fallback role. Selection is reconciled against Smithers' durable -attempt metadata and the sealed task chain; it is never inferred from the retry -number or token model. Failed primaries therefore remain visible even when a -later fallback produces the accepted output. +primary-or-fallback role when Smithers' durable attempt metadata identifies a +rung of the sealed task chain; selection is never inferred from the retry number +or token model. Failed primaries therefore remain visible even when a later +fallback produces the accepted output. An attempt that fails before Smithers +selects a rung ran no model and is not recorded, unless a reset supersedes it +first (see below). + +Each attempt is identified by its terminal Smithers event and is recorded once: +later synchronization never re-derives or rewrites it, even when the node's +status changes afterwards. Finished, failed, timed-out, and cancelled attempts +are recorded; a cancellation has outcome and category `canceled` and the +Smithers cancellation reason as its message. A terminal event with no started +attempt in the same Smithers activation, or stamped before that attempt +started, is skipped. A finished attempt whose node then fails, for example +because the verifier or artifact gates reject its output, is recorded as failed +with category `invalid-output` for findings validation and +`artifact-validation` otherwise. +`resume --retry-failed` and `--reset-node` restart Smithers' attempt numbering, +after which Smithers' attempt row describes only the replacement. A failed, +timed-out, or cancelled attempt that such a reset superseded before any +synchronization recorded it is therefore recorded from its events without the +agent block. That includes a pre-agent failure, which then counts toward the +node's `retry_count` although no model ran. A superseded finished attempt that +was not recorded before the reset is not recorded, because the node's output +manifest now belongs to the replacement. Attempt summaries and retry counts are derived from this ledger. Replaying a known transition does not append it again, so resume, replay, checkpoint continuation, and controller takeover preserve prior lifecycle history. Reused work points to its source attempt and is reported separately from executed work. The ledger stores typed failure categories but never raw diagnostics, inputs, -outputs, or configuration. +outputs, or configuration. Ledger bookkeeping does not block synchronization: +when Smithers attempt detail is unavailable or an append fails, synchronization +reports a warning, still reconciles node and run status, and retries on the next +pass. Synchronization reads each Smithers event stream with +`smithers events --limit 100000`, the CLI maximum, which returns the oldest +events first; a stream that returns exactly that many events is reported as +`WORKFLOW_EVENTS_TRUNCATED`, because any later attempts or usage cannot be read. For terminal report producers, `report.json#run_metadata.agent_execution` contains the full planned attempt chain, the attempts that failed before the @@ -537,8 +574,9 @@ artifacts/final-report/report.json `report.json` must satisfy `ultrafuzz/report@3` with the exact `ultrafuzz.report.v3` version literal. After a run stops, the runtime can format -that report and attach whole-run completion information without changing the -agent's files. Verified publications use: +that report, attach whole-run completion information, and restate the run +summary's elapsed time and accounting without changing the agent's files. +Verified publications use: ```text review/runtime-report//report.json @@ -738,14 +776,23 @@ Blockers: Repeat blocker and evidence rows in artifact order. Runtime publication compares this section with the typed handoff and rejects missing, duplicated, reordered, -or bare coverage scores. Raw `covg-eval` output is for iteration only and -defines neither published declaration-completeness view. - -Current-run `report.md` contains concise links to `THREAT_MODEL.md`, -`threat-model.json`, and `goal-plan.json`, plus source-node provenance for each -production issue. Detailed threat analysis stays in the dedicated threat-model -artifacts and is not duplicated into the report. `report.json` preserves the -same `source_nodes` arrays. +or contradicting scoped scores, and it warns about coverage scores that name no +exact scope. Raw `covg-eval` output is for iteration only and defines neither +published declaration-completeness view. + +A coverage score that names no exact declaration-completeness scope, whether in +`report.md`, the coverage producer's Markdown, or `report.json` text, does not +fail publication, although text that exceeds the 2,048-candidate scan limit +still does. When no coverage producer was planned or admitted, +`report.json.coverage_evidence` or a `report.md` score that names an exact +scope fails the final report. + +Current-run `report.md` contains source-node provenance for each production +issue and does not link to other run files. Detailed threat analysis stays in +the dedicated threat-model artifacts and is not duplicated into the report. +`report.json` preserves the same `source_nodes` arrays. Inline link and image +syntax inside report prose, including prose preserved byte-for-byte from +upstream findings, renders as literal text. When workflow usage data is available, run metadata includes `accounting.cumulative.tokens_used` and @@ -755,13 +802,20 @@ available cumulative values into the markdown run summary and into persisted estimate is partial because some token usage did not have pricing data. -If cumulative metadata has not synchronized when the final-report producer -starts, its live Smithers fallback is a snapshot through that producer's start. -It includes earlier attempts but cannot include the producer's own eventual -duration, model fallback, tokens, or cost. A terminal presentation of an existing -verified agent report preserves those accounting values. Report v3 has no -metric-scope field, so use -`ultrafuzz stats` after terminal synchronization for closed-run accounting. +The final-report producer receives its run summary when its task starts: from +cumulative metadata when it has synchronized, otherwise from a live Smithers +fallback. Either way it is a snapshot through that producer's start. It includes +earlier attempts but cannot include the producer's own eventual duration, model +fallback, tokens, or cost, and the agent's `report.json` and `report.md` keep +that snapshot. Runtime presentations (the verified terminal publication and +unchecked reports) restate the run summary instead: elapsed time from +`run.json#created_at` to `state.json#finished_at`, and models, tokens, +estimated spend, and `partial_pricing` from the current +`accounting.cumulative`. Tokens, estimated spend, and `partial_pricing` are +restated together whenever `accounting.cumulative` records a token count, so a +whole-run spend recorded as `unavailable` stays `unavailable` instead of showing +the agent's report-start figure. Otherwise, a value those records lack keeps the +agent's copy. Use `ultrafuzz stats` for the full accounting breakdown. `accounting.segments` publishes one rollup per checkpoint generation, and `accounting.current` identifies the latest segment. Each segment retains every @@ -769,7 +823,15 @@ source event sequence as audit evidence, but accounting uses only the latest cumulative usage snapshot for each workflow attempt. `accounting.cumulative` combines those canonical attempt snapshots with any source-run lineage. `accounting.checkpoint` records the raw ledger position used by the durable -metadata snapshot. Usage and pricing completeness are reported independently +metadata snapshot. The accounting block is a cache that every synchronization +rebuilds from `usage.jsonl`, and a usage row is recorded once and never +re-derived, so a synchronization interrupted between the usage append and the +`run.json` write is repaired by the next one. A failed accounting pass is +reported as a `WORKFLOW_ACCOUNTING_FAILED` warning and does not block run status. +An invalid usage field in an unrecorded `TokenUsageReported` event fails each +later accounting pass this way, so no further usage is recorded for the run +while node and run status keep reconciling. +Usage and pricing completeness are reported independently through `usage_complete`/`usage_incomplete_reasons` and `pricing_complete`/`pricing_incomplete_reasons`. @@ -796,7 +858,9 @@ a usable cost. Kimi-family models are priced from the pinned Moonshot provider entry, while DeepSeek-family models are priced from the pinned first-party DeepSeek entry. Either family stays listed in `pricing_catalog.unresolved_models` when its first-party entry is absent rather -than borrowing a same-named rate from another provider. +than borrowing a same-named rate from another provider. A model that a fetched +catalog does not list stays unresolved without another catalog download; only +an unavailable catalog is retried on a later synchronization. The final report is a review artifact. It is not an automatic vulnerability submission, repository mutation, or patch application. @@ -857,7 +921,6 @@ runs.jsonl scores.jsonl summary.json summary.md -telemetry/ ``` `eval.json` records the resolved suite plus candidate and benchmark lineage, @@ -892,10 +955,7 @@ available. `ultrafuzz eval score` writes per-row scores to `scores.jsonl` and the variant ranking plus scoring lineage to `summary.json`, including the effective deterministic or optional-judge mode. -`telemetry/` holds durable per-row telemetry cursors with byte offsets, event -deduplication state, and artifact hashes for the local observer loop. Historical -publication cursor documents remain readable, but there is no external -publication command. The underlying Ultrafuzz runs live inside each target +The underlying Ultrafuzz runs live inside each target checkout, not under the eval project; eval artifacts reference them by run ID. Grading and these artifacts do not depend on a reporting service. See [Eval Suites](evals.md). diff --git a/docs/reference/cli.md b/docs/reference/cli.md index 0e42102b2..59097b75a 100644 --- a/docs/reference/cli.md +++ b/docs/reference/cli.md @@ -104,16 +104,19 @@ ultrafuzz.toml .ultrafuzz/cache/ ``` -Without `--force`, existing config, topology, prompts, reference catalogs, and -agent adapters are preserved except for recognized stock migrations. `init` +Without `--force`, existing config, topology, prompts, and reference catalogs +are preserved. The `.smithers/agents` adapter closure is not project-owned: +planning accepts only the byte-exact packaged copy, so every `init` rewrites +each adapter that differs from the current templates and leaves matching ones +untouched. Refreshing the adapters after an upgrade therefore needs no +`--force`. A symbolic link, hard link, or special file at an adapter path makes +`init` fail with `INIT_PATH_UNSAFE` instead of being followed or written +through. `init` also migrates a recognized generated `smithers-orchestrator@0.32.0` manifest or a `smthrs@0.34.0` manifest to the current `smthrs@0.35.0` pin. The 0.34.0 migration -requires the rest of the manifest to match the current generated document. -During a recognized manifest migration, known byte-identical prior stock -adapters are also upgraded; customized or unrecognized adapters remain -preserved. Use `--force` to replace existing generated files with the current -templates. `init` emits an actionable diagnostic when a preserved adapter -requires manual review. +requires the rest of the manifest to match the current generated document. Use +`--force` to replace the other existing generated files with the current +templates. ## Validate @@ -124,7 +127,10 @@ ultrafuzz validate [--project ] [--audit-profile ] \ Validation covers typed TOML config, `.ultrafuzz/topology.yml`, project prompt copies, safe paths, reference nodes, agent references, and trusted local -execution posture. It does not launch agents. +execution posture. It does not launch agents. A project prompt that differs +from the built-in prompt at the same path sets the prompts posture to `warn` +with one `PROMPT_DIFFERS_FROM_BUILT_IN` warning per file; warnings do not fail +validation. ## JSON Validate @@ -161,8 +167,9 @@ package-local registries. A registered schema whose filename or bytes differ from its pinned entry is a setup failure. Successful JSON output reports whether the schema was registered plus its fragment-free ID, schema SHA-256, owning package's bundle SHA-256, validator build identity, and the artifact -SHA-256. These identities bind the producer command to the later host check; -they are not a mutable validation receipt. +SHA-256. The schema identities bind the producer command to the later host +check, and the validator build is reported as provenance only; none of them is +a mutable validation receipt. For schema-backed producer tasks, Ultrafuzz places a run-owned trusted launcher before target-controlled `PATH` entries and validates a real known-valid fixture @@ -293,7 +300,7 @@ ultrafuzz node \ [--project ] \ [--json] ultrafuzz resume [--project ] [--max-concurrency ] \ - [--reset-node ] [--refresh-controller] [--json] + [--reset-node ] [--retry-failed] [--refresh-controller] [--json] ultrafuzz replay [--project ] [--json] ultrafuzz fork \ [--project ] \ @@ -316,6 +323,18 @@ the linked workflow status when it is available; its JSON output retains both `workflow_status` and `ultrafuzz_status` so lifecycle divergence remains explicit. +A launch that fails after planning has created its run directory, but before +its workflow controls are sealed (while planning, compiling the workflow, or +checking its task manifest), records the run as `failed` with its original +error. `status`, `resume`, and the other commands that read the linked +workflow then report that error as `RUN_LAUNCH_FAILED`. Such a run never +reached Smithers and cannot be resumed; fix the cause and start a new run. A +launch that fails after sealing is also recorded as `failed`, with its error in +`events.jsonl`, but those commands do not report it as `RUN_LAUNCH_FAILED`. A +goal-planning topology without its vulnerability-database reference, and a +missing or invalid reference cache (run `ultrafuzz references sync`), are +rejected before the run directory is created. + `resume` delegates continuation to Smithers with the same Ultrafuzz and Smithers run IDs and automatically accepts changed workflow source. Control seals, link journals, controller generations, graph fingerprints, current @@ -323,15 +342,34 @@ schema bindings, and metadata projections remain provenance for inspection; they are not resume authorization. Smithers decides which finished rows can be reused and which newly rendered or unfinished tasks run. Ultrafuzz does not rewrite historical artifacts or automatically reset, replay, timetravel, or -fork completed work. +fork completed work. When the run's `smithers/resolved-config.json` parses as +the current resolved-config schema, agent adapters in the continued workflow +read the run's `smithers/execution-config.toml` (launch gave them a copy of the +same file); for a run whose resolved config does not parse, resume sets no +`ULTRAFUZZ_CONFIG_PATH`. If resume cannot prune stale task-worktree +registrations, it reports a `WORKFLOW_WORKTREE_REPAIR_FAILED` warning and +continues. + +A run that ends `failed` with no failed durable node was stopped by something +no durable node owns: a run-level workflow runner error, such as an exception +thrown while rendering the workflow, or a failed workflow task outside the +durable graph. `status` reports it as `WORKFLOW_TERMINAL_WITHOUT_FAILED_NODE`, +names any failed workflow tasks, and appends the runner's run-level error to +the message. For a run-level error, fix that cause, then `resume` the same run; +it continues from the tasks that already finished. While the cause persists, +the resume either fails with `WORKFLOW_LIFECYCLE_FAILED` or the run fails +again. `resume --refresh-controller` first renders the currently installed Ultrafuzz controller and stock adapters beside the historical source, then delegates to that same Smithers run. It does not publish or authenticate a historical -controller generation. Refresh rejects an actively owned workflow. Because -Smithers admits changed workflow source, replay determinism is the operator's -responsibility; inspect the retained source and Smithers workflow hash when -auditing a continuation. +controller generation. The refreshed workflow rebinds each declared output to +the installed schema bundle but keeps its recorded contract digest and validator +build, so after a rebuild that changed only the validator build its verification +markers still match the run's sealed plan. Refresh rejects an actively owned +workflow. Because Smithers admits changed workflow source, replay determinism is +the operator's responsibility; inspect the retained source and Smithers workflow +hash when auditing a continuation. `status` human output is watch-friendly: @@ -458,13 +496,18 @@ an available agent-written report. `pause` requests a graceful stop: no new tasks are scheduled, in-flight tasks finish, and the run settles in the resumable `paused` state. `resume` reports -`submitted: false` instead of launching a duplicate continuation when the linked -workflow is still in an active state (running, in-progress, started, queued, -retrying, or waiting). `resume --reset-node` retries one failed workflow node and -its dependents in the same linked run; the applied reset is recorded so retrying -the command after a failed continuation resumes the already-reset run instead of -repeating the reset. `fork` may start from a checkpoint frame and may reset one -workflow node before starting the fork. +`submitted: false` (text output `Run already active`) instead of launching a +duplicate continuation when the linked workflow is still active (its Smithers +run state is `running`, `recovering`, or one of the `waiting-*` states), and +leaves the run's recorded state and workflow deadline unchanged. A run still +finishing its in-flight tasks after `pause` is still active; resume it again +once `status` reports `paused`. Smithers also reports a run as `running` for up +to 30 seconds after its controller process exits (its heartbeat window), so +resume such a run again after that. `resume --reset-node` retries one failed +workflow node and its dependents in the same linked run; the applied reset is +recorded so retrying the command after a failed continuation resumes the +already-reset run instead of repeating the reset. `fork` may start from a +checkpoint frame and may reset one workflow node before starting the fork. Every command in this section takes an Ultrafuzz run ID and resolves the linked workflow run from existing product evidence; none of them require the @@ -480,6 +523,24 @@ failing. Both outcomes append distinct product events. Failures use the stable `WORKFLOW_CANCEL_FAILED` diagnostic. +`pause` and `cancel` read run evidence the way `status` does, without the +workflow control lock. A run whose sealed control documents diverged, for +example a hand-patched published workflow or a planned graph that no longer +matches the current build's artifact contracts after a rebuild, can therefore +still be paused or cancelled; `status` reports the divergence. They start the +workflow runner from the run's published execution snapshot, so, like +`status`, they still refuse a run whose sealed execution files changed: the +files the control seal lists in that snapshot, such as the run plan, prompts, +agent adapters, and the runtime packages and their dependencies. + +Because they take no lock, `pause` and `cancel` issued while a launch is still +preparing fail without changing the run; retry once `ultrafuzz run` has +returned. They also do not reconcile a `replay` or `fork` that was interrupted +while linking its new workflow run, so they act on the workflow run it +replaced, or refuse. Run `ultrafuzz why ` before pausing or cancelling +such a run: it reconciles that link, even when it then reports a diverged +control document. + `why` returns a deterministic diagnosis: a summary, the current node, and typed blockers with `kind`, `node_id`, `iteration`, `reason`, `unblocker`, `waiting_since`, `attempt`, and `max_attempts`. Blocker kinds are @@ -533,31 +594,33 @@ non-launching configuration contract unchanged. Doctor reports: - config, topology, prompt, and reference validation posture; - required topology backends, toolchain, and configured agent executable availability in the configured execution environment (the local `PATH` for - local runs or a transient probe of the provider image for cloud runs); + local runs or a transient probe of the provider image for cloud runs). Only + the executables of agents the selected topology can dispatch to, including + its `[retry] agents` fallbacks, are required; other configured profiles' + executables are listed as not required; - the bundled workflow engine version, the version the generated project requires, and the installed project-local version and bin target; - npm's latest published stable engine version when the registry check is available; -- whether the installed dependency layout passes Ultrafuzz's exact - manifest and path validation; -- whether required compatibility patches, or their upstream replacements, are - present. A source carrying neither the patch nor the shape Ultrafuzz patches - is reported as modified or incompatible, because the next run fails in that - state. +- whether the project-local dependency layout passes Ultrafuzz's exact + manifest and path validation, and the posture of each compatibility patch in + it. Both are informational: a launch installs, patches, and seals its own + operator-owned controller; +- the OS temporary directory, where launch and resume install that controller: + its free space and how many `ultrafuzz-controller-*` directories it holds, + with their total size. Doctor warns when the directory is a RAM-backed tmpfs + or has less than 2 GiB free, and never removes those directories, because a + native resume keeps its controller there for the detached engine. Doctor does not create project run state or install, upgrade, or repair local dependencies. For cloud execution, checking required commands may create the configured provider app on first use and uses a transient sandbox so the probe runs inside the same image as workflow nodes. -Diagnostics are stable: `DOCTOR_TOOLCHAIN_MISSING`, -`DOCTOR_TOOLCHAIN_PROBE_FAILED`, -`DOCTOR_WORKFLOW_ENGINE_MISSING`, -`DOCTOR_WORKFLOW_ENGINE_VERSION_MISMATCH`, -`DOCTOR_WORKFLOW_ENGINE_LAYOUT_INVALID`, -`DOCTOR_WORKFLOW_ENGINE_PATCHES_PENDING`, -`DOCTOR_WORKFLOW_ENGINE_PATCHES_INCOMPATIBLE`, -`DOCTOR_WORKFLOW_ENGINE_OUTDATED`, and `DOCTOR_REGISTRY_UNAVAILABLE`. A registry or network failure produces a +Diagnostics are stable: `DOCTOR_AGENT_CREDENTIAL_MISSING`, +`DOCTOR_TOOLCHAIN_MISSING`, `DOCTOR_TOOLCHAIN_PROBE_FAILED`, +`DOCTOR_WORKFLOW_ENGINE_OUTDATED`, `DOCTOR_REGISTRY_UNAVAILABLE`, and +`DOCTOR_TEMPORARY_DIRECTORY_CONSTRAINED`. A registry or network failure produces a warning and an `unknown` latest version instead of failing an otherwise valid offline project. Doctor never installs, mutates, or upgrades dependencies. @@ -744,8 +807,8 @@ directory per target id under `--target-root`) against the pinned git refs; `--skip-target-validation` skips that check. `run` launches Ultrafuzz runs for matrix rows (all rows, or a `--row` -selection) and polls them to a terminal state. Reports and telemetry remain -in local run artifacts. `--no-watch` launches detached without polling. +selection) and polls them to a terminal state. Reports remain in local run +artifacts. `--no-watch` launches detached without polling. `status` reads the eval matrix, its latest `runs.jsonl` records, and each linked durable `state.json` without synchronizing or changing workflow state. @@ -797,5 +860,4 @@ candidate repository, immutable source artifact, and publication URL flags. performs no writes. Eval artifacts are written under `.ultrafuzz/evals/runs//`. See -[Eval Suites](evals.md) for configuration, architecture, and telemetry policy -details. +[Eval Suites](evals.md) for configuration and architecture details. diff --git a/docs/reference/configuration.md b/docs/reference/configuration.md index 1095c7155..61ec63a27 100644 --- a/docs/reference/configuration.md +++ b/docs/reference/configuration.md @@ -139,18 +139,21 @@ resume. `workflow_deadline_seconds` is not a guaranteed wall-clock limit. Ultrafuzz records `workflow_deadline_at` in run state when the run is created, and again -from each `resume`, `replay`, or `fork`, but nothing enforces it on a timer: an +from each `replay`, `fork`, or `resume` that starts a controller (not one that +finds the run still active), but nothing enforces it on a timer: an unattended run keeps executing, and incurring provider cost, past its deadline. The deadline is checked only when a command synchronizes the run: `ultrafuzz status` (including each `--watch` poll), `inspect`, `why`, and `stats`, plus the dashboard's inspect action and the eval runner's poll loop. `ultrafuzz run` -does not check it. If the run is not yet terminal (running, pending, or paused) -at the first synchronization after the deadline, that synchronization requests -cancellation and, when the request succeeds, marks the run `timed-out` and -appends a `workflow-deadline-exceeded` event. A failed request is reported as a -`WORKFLOW_DEADLINE_CANCEL_FAILED` error and leaves the run active; a -synchronization that fails or is skipped (for example -`WORKFLOW_STATE_SYNC_SKIPPED`) does not check the deadline at all. A run that +does not check it. If the run is running or pending at the first +synchronization after the deadline, that synchronization requests cancellation +and, when the request succeeds, marks the run `timed-out` and appends a +`workflow-deadline-exceeded` event. A paused run is not cancelled: it executes +nothing, and resuming it records a new deadline. A failed request is reported as +a `WORKFLOW_DEADLINE_CANCEL_FAILED` warning and leaves the run active, and the +next synchronization requests cancellation again; a synchronization that fails +or is skipped (for example `WORKFLOW_STATE_SYNC_SKIPPED` or +`WORKFLOW_SYNC_IN_PROGRESS`) does not check the deadline at all. A run that finished first keeps its terminal outcome, with no timeout record. To bound an unattended run, run `ultrafuzz status ` periodically (for example from cron) and act on its warnings, or cancel it with `ultrafuzz cancel `. @@ -174,8 +177,7 @@ Every submitted workflow starts a run-scoped recovery supervisor. The supervisor renews controller ownership through runner heartbeats and uses an atomic claim before taking over expired ownership, so completed work is not resubmitted. The workflow deadline is separate from per-node timeouts and is -checked whenever run state is synchronized by status, inspect, reporting, or -eval watchers. +checked only when run state is synchronized, as described above. ## Execution @@ -387,6 +389,7 @@ is local and remains the default. | `ULTRAFUZZ_PRICING_CATALOG_URL` | Live model-pricing catalog URL, or `disabled`, `none`, or `off`. | | `ULTRAFUZZ_PRICING_TIMEOUT_MS` | Positive catalog request timeout in milliseconds, capped at 60 seconds. | | `ULTRAFUZZ_OBSERVATION_SYNC_TIMEOUT_MS` | Optional deadline in milliseconds for the run-state refresh before `status`, `inspect`, `why`, and `stats`. Unset means no deadline (observers wait for full synchronization); a positive value bounds it, capped at 60000; `0` or `off` is the same as unset. | +| `ULTRAFUZZ_RUNNER_QUERY_TIMEOUT_MS` | Timeout in milliseconds for each read-only runner query (`inspect`, `events`, `node`, `status`, `why`, `ps`, `timeline`, `snapshots`); one that exceeds it fails. Defaults to 120000, capped at 600000; `0`, `off`, or any invalid value means the default. | Boolean values accept `1`, `true`, `yes`, `on`, `0`, `false`, `no`, and `off`. @@ -400,6 +403,20 @@ the same as unset. When a configured deadline passes, the command reports a `WORKFLOW_SYNC_DEADLINE_EXCEEDED` warning and continues with the local run state, whose node states, attempt ledgers, and usage counts may then be stale. +A command or eval poll that finds another synchronization of the same run in +progress skips its own pass instead of running alongside it, reports +`WORKFLOW_SYNC_IN_PROGRESS`, and continues with the local run state. The lock +behind this lives in `.workflow-sync.lock` in the run directory; one left behind +by a killed process is taken over five minutes after its holder last refreshed +it. A pass that cannot create the lock, for example in a read-only run +directory, reports `WORKFLOW_SYNC_LOCK_FAILED` and writes nothing. During its +refresh, `status` reports a failed or malformed runner `inspect` or `events` +query, a lock it cannot take, and an unexpected synchronization error as +warnings, so it still shows the workflow runner's health and `--watch` keeps +polling. When nothing but the observation time changed, +`status` and any synchronization of a finished run leave `state.json` untouched; +any other synchronization of a live run still renews its controller lease. + Custom pricing catalogs must use HTTPS without credentials, query parameters, or fragments and must resolve entirely to public addresses. The validated DNS address is pinned for the request, redirects are rejected, and response bodies diff --git a/docs/reference/development.md b/docs/reference/development.md index 0c441735a..4a7e68158 100644 --- a/docs/reference/development.md +++ b/docs/reference/development.md @@ -40,19 +40,19 @@ The root CI script runs format check, lint, build, and release validation: pnpm -w run ci ``` -Pull requests run CI policy checks, dependency policy, formatting, lint, the -workspace build, and a curated runtime smoke suite. The smoke suite reuses the -built workspace and covers runtime sharding, workflow controls, source revision -binding, generated workflow input, and representative initialization, -validation, planning, and workflow-compilation behavior. Feature branches are -validated only by the pull-request event, avoiding a duplicate push run. - -Pull requests also require all five runtime validation lanes: one supporting -lane, including the Bun adapter contracts, and four deterministic integration -shards. Together these run the full runtime suite before merge. Pushes to `main` -and manual workflow dispatches run all eight release lanes, adding package, -CLI, and benchmark-history/typecheck checks, with at most eight jobs in parallel. -Their results are recorded in stable gate order in the JSON report. +Every CI run, pull requests included, runs the build gates (CI policy checks, +formatting, lint, dead-code checks, the workspace build, strict lint of changed +lines, bundle budgets, and dependency policy) and all nine release validation +lanes: package gates, one runtime supporting lane that includes the Bun adapter +contracts, four runtime integration shards, the CLI suite, the `cli-e2e` lane, +which runs one campaign end to end (see below), and benchmark history with the +workspace typecheck. The lanes do not wait for the build gates and run at most +eight at a time. They run their tests under `eatmydata`, which turns `fsync` +into a no-op in the test processes but not in the Smithers engine processes +those tests launch. Feature branches are validated only by the pull-request +event, avoiding a duplicate push run. A newer push to a pull request cancels +that pull request's older run; a push to `main` never cancels another run. On +`main`, the lane results are merged into one JSON report in stable gate order. ## Package Checks @@ -69,6 +69,17 @@ pnpm --filter @ultrafuzz/cli test pnpm --filter @ultrafuzz/modal test ``` +`pnpm --filter @ultrafuzz/cli test:e2e` runs the end-to-end campaign test in +`packages/cli/test/e2e/`. It drives `init`, `run`, `resume`, `status`, `stats`, +`report`, and `events` as separate CLI processes, and the generated workflow +runs on the pinned Smithers engine under Bun, with a stub `codex` executable in +place of the model. It SIGKILLs the detached controller while one node is +running, resumes the run, and checks that it succeeds with a verified report and +that no finished task started again. It needs Linux, Bun, Git, and access to +the npm registry, because `run` and `resume` install the pinned engine from npm +as they do for any campaign. On SIGINT or SIGTERM, the test kills the detached +campaign and deletes its fixture, about 1 GB, before it exits. + Package-local `typecheck` and `test` scripts may build direct workspace dependencies first because package exports point at `dist/**`. diff --git a/docs/reference/evals.md b/docs/reference/evals.md index c31b25105..919e90e39 100644 --- a/docs/reference/evals.md +++ b/docs/reference/evals.md @@ -35,14 +35,14 @@ provider = "none" See `.ultrafuzz/evals/bug-finding.yml` for the default suite. It defines model profiles, targets (repo/ref/ground truth/sensitivity), variants, trial counts, -grading metrics, and the `reporting:` telemetry policy. Nothing in the YAML -names a provider, an endpoint, or an env var. +grading metrics, and the `reporting:` block. Nothing in the YAML names a +provider, an endpoint, or an env var. -The generic telemetry API retains `reporting.artifacts` policies for explicit -programmatic observers. Private targets default to `manifest-only`; payload -access requires an explicit allowlist and is checked against path containment, -regular-file status, size, and SHA-256 digest. This policy does not install a -reporter or cause the built-in CLI to upload artifacts. +`reporting.node_telemetry` decides whether `ultrafuzz eval run` watches +launched rows by default; `--no-watch` overrides it. It and +`reporting.heartbeat_interval_seconds` remain inputs to the execution-policy +fingerprint. `reporting.experiment_prefix` and `reporting.artifacts` are still +validated, but no eval behaviour depends on them. ### Suite contract and workflow input @@ -220,27 +220,17 @@ graph-inconsistent execution ledgers fail closed as non-comparable. ## Architecture -Reporting is **event-sourced from the run journal**, never wired inline into -the workflow runner: - -- `packages/evals/src/reporter.ts` defines the generic `EvalReporter` observer - interface. Ultrafuzz owns the local loop and writes `matrix.json`, - `runs.jsonl`, `scores.jsonl`, and `summary.json`. No built-in external - implementation is registered; a TOML connection profile cannot add one. -- `packages/evals/src/node-telemetry.ts` is the pump: a cursor over - `events.jsonl` + `state.json` + artifact manifests, driven from the eval - driver's poll loop. The cursor (byte offset + `event_id` dedup ring + - uploaded-artifact hashes) is reloaded and persisted under a per-cursor lease. - Delivery is at-least-once: callbacks happen before the durable cursor commit, - so a crash or cursor persistence failure can replay a callback. Reporters must - make those callbacks idempotent with the stable `idempotencyKey` supplied on - every event envelope and artifact upload. Event keys are derived from the eval - row plus journal `event_id`; artifact keys are derived from row, node, relative - path, and SHA-256. Exhausted provider delivery retries degrade to warnings; - cursor lock, validation, and persistence failures stop the drain. -- Heartbeat liveness is bounded by the sync poll cadence: state transitions - and partial artifacts appear within one poll interval. That is the correct - trade for a detached orchestrator. +The eval driver owns the local loop and writes `matrix.json`, `runs.jsonl`, +`scores.jsonl`, and `summary.json`; nothing is wired into the workflow runner. +`ultrafuzz eval run` launches each row as a detached Ultrafuzz run. When it +watches a row, it repeats three steps until the run's durable `state.json` is +terminal or the watch deadline passes: synchronize the run, read `state.json`, +and sleep for the poll interval. It reads `state.json` once more before +recording the row, so a run that another command synchronized to a terminal +state during the last sleep is not recorded as timed out. A failed +synchronization is counted and recorded on the row as `EVAL_ROW_SYNC_FAILED`; +it does not end the watch. +There is no reporter or telemetry-export interface. ## Versioned lineage @@ -281,7 +271,7 @@ differences that were waived. ```bash ultrafuzz eval plan # validate config + suite, print the matrix -ultrafuzz eval run # launch rows, poll to terminal state, retain local telemetry +ultrafuzz eval run # launch rows and poll them to a terminal state ultrafuzz eval status # observe every row's durable node progress and ETA ultrafuzz eval score # grade reports against ground truth (optional --llm-judge) ultrafuzz eval report # show the scored variant ranking @@ -506,7 +496,7 @@ LLM result cannot downgrade a deterministic true positive. Full per-command flags are in the [CLI reference](cli.md#eval). Local eval artifacts (`eval.json`, `matrix.json`, `runs.jsonl`, `scores.jsonl`, -`summary.json`, `summary.md`, telemetry cursors) are documented in +`summary.json`, `summary.md`) are documented in [Run Artifacts and Reports](artifacts-reports.md#eval-run-artifacts). For a task-oriented walkthrough, see [Run Eval Suites](../how-to/run-evals.md). diff --git a/docs/reference/prompt-variables.md b/docs/reference/prompt-variables.md index 32d78ba75..e642538b0 100644 --- a/docs/reference/prompt-variables.md +++ b/docs/reference/prompt-variables.md @@ -174,6 +174,11 @@ bounded, item-scoped `replacements` map. Namespaced keys such as `{{liquidation:overdue}}` resolve only from that item. A value such as `{{item.goal_prompt}}` may retain those placeholders for the bounded nested replacement pass; unresolved, cyclic, non-scalar, or non-item references fail. +That failure happens while the workflow renders, so it stops the whole run +rather than one node. For `ultrafuzz/goal-plan@1`, `goal-plan`'s own +verification therefore rejects `{{` anywhere in a replacement value, so a bad +label fails that verification instead. `ultrafuzz json validate` and +`ultrafuzz artifact validate` do not run this check. ## Output Contract diff --git a/docs/reference/topology-yaml.md b/docs/reference/topology-yaml.md index 65708ce28..9eabe3b6c 100644 --- a/docs/reference/topology-yaml.md +++ b/docs/reference/topology-yaml.md @@ -225,6 +225,9 @@ Reference nodes must: - Mark exactly one non-manifest output as primary. - Avoid `prompt`, `role`, and `model_profiles`. +Reference nodes are materialized when the run is planned and never run as +agent tasks, so `timeout_seconds` has no effect on their execution. + Downstream prompts can consume the normalized Markdown primary artifact with: ```md @@ -346,9 +349,16 @@ Generated children use the same global concurrency scheduler as static nodes. exceeding it fails explicitly and never truncates the source array. The first successful expansion is persisted under -`dynamic-expansions/.json`. Resume reuses that exact manifest and -rejects changes to its source bytes, prompt template, topology contract, or -dynamic-node limit instead of silently changing the graph. +`dynamic-expansions/.json`, and from then on that manifest alone +decides the group's generated nodes. Resume reuses it and rejects changes to its +prompt template, topology contract, or dynamic-node limit instead of silently +changing the graph. The source artifact is read only to create the manifest, so +a source node that runs again, for example after a reset, leaves the published +fan-out unchanged. The exception is `resume --retry-failed` for a source whose +verifier failed: it moves the published manifests to +`dynamic-expansion-history/` before the source runs again (and refuses if +another source published any of them), so the group expands again from the new +output. ## Model Fan-Out @@ -420,8 +430,10 @@ the planned output: - validator build identity. The expanded graph, run state, verification marker, and -`artifact-manifest.json` carry the same identity. The host rejects a missing, -partial, stale, or mismatched binding before publication. +`artifact-manifest.json` carry the same identity. The host rejects a missing or +partial binding, and an artifact its planned schema content rejects, before +publication. The validator build identity is recorded as provenance and is not +compared with the build doing the checking. Topology YAML remains version `2`; the persisted expanded graph uses `graphVersion: "4"` and schema ID diff --git a/docs/schemas.md b/docs/schemas.md index c8e8caedc..6524e24f7 100644 --- a/docs/schemas.md +++ b/docs/schemas.md @@ -22,8 +22,6 @@ Schema IDs are stable, fragment-free URNs such as: - `urn:ultrafuzz:schema:evals:benchmark-lanes:2` - `urn:ultrafuzz:schema:evals:ground-truth:1` - `urn:ultrafuzz:schema:evals:history:2` -- `urn:ultrafuzz:schema:evals:history-automatic-publication-plan:1` -- `urn:ultrafuzz:schema:evals:history-publication-generation:1` - `urn:ultrafuzz:schema:evals:run-record:3` - `urn:ultrafuzz:schema:evals:recovery-equivalence:1` - `urn:ultrafuzz:schema:evals:status:1` @@ -66,25 +64,37 @@ ultrafuzz json validate \ Use repeatable `--ref` flags only for explicitly supplied local dependencies. Bundled sibling schemas resolve offline without flags. The CLI and host use the same non-mutating parser, schema registry, Ajv configuration, resource limits, -schema digest, bundle digest, and validator-build identity. +schema digest, and bundle digest, and each reports its validator-build identity. Every planned JSON output persists the registered schema filename, `$id`, schema SHA-256, owning package's schema-bundle SHA-256, and validator build. The expanded graph, run state, `ultrafuzz.artifact-verification.v2` marker, and -`ultrafuzz.artifact-manifest.v3` repeat that binding. A missing, partial, stale, -or mismatched identity is a host setup/verification failure even when the JSON -would match a different schema with the same general shape. +`ultrafuzz.artifact-manifest.v3` repeat that binding. The host validates each +artifact against the schema content that binding names: the installed schemas +when their bundle digest is the planned one, otherwise the bundle sealed in the +run's execution snapshot. A missing or partial binding, or schema content that +neither holds, is a host verification failure even when the JSON would match a +different schema with the same general shape. The validator build is +provenance: graph reads, host artifact gates, and the validator preflight do not +compare it with the build doing the checking. Schema-backed producers receive a run-owned trusted launcher ahead of target-controlled `PATH` entries. Local and Modal environments use that launcher -to validate a real known-valid fixture and compare the returned schema, bundle, -and validator-build identity before model work. Local launchers resolve only a -verified content-addressed snapshot of the CLI and every transitive package, so +to validate a real known-valid fixture and compare the returned schema and +bundle identity before model work; the returned validator build is provenance. +Local launchers resolve only a verified content-addressed snapshot of the CLI +and every transitive package, so a working-tree rebuild cannot change an active run. Ambient Node loader/search variables are removed and both ESM and CommonJS module resolution must stay inside that snapshot; document reads are unaffected. A path lookup alone is not a preflight. Ordinary resume now delegates continuation to Smithers instead of -using the historical launcher or closure as an authorization gate. A current +using the historical launcher or closure as an authorization gate. When resume +cannot re-verify the launcher, it reports a `WORKFLOW_TRUSTED_CLI_UNVERIFIED` +warning and, if `/trusted-bin/ultrafuzz` exists, keeps it first on `PATH` +rather than letting tasks reach another `ultrafuzz`. That launcher still +verifies its closure before every dispatch, so if its metadata or closure is +damaged, or the Node binary it names is gone, each task's validator preflight +fails; `resume --refresh-controller` does not repair such a launcher. A current controller refresh publishes a new controller path without rewriting the historical closure. diff --git a/docs/security.md b/docs/security.md index 9b5c134dc..3c07a5953 100644 --- a/docs/security.md +++ b/docs/security.md @@ -93,6 +93,28 @@ remain fail-closed pending the separate R-26 disclosure authorization. Public Modal runs record `cloud:modal`. These controls are not a sandbox or egress filter: YOLO agents remain unrestricted. +A model destination is `model:`, or `model:-route-` +when route input is present. The digest covers: + +- every non-credential environment variable with the agent's provider prefix + (`ANTHROPIC_` and `CLAUDE_CODE_USE_` for Claude, `OPENAI_` and + `AZURE_OPENAI_` for Codex, `KIMI_` and `MOONSHOT_` for Kimi), including ones + that do not route traffic, such as `ANTHROPIC_LOG`; +- Claude's `AWS_`, `GOOGLE_`/`CLOUD_ML_`, and `AZURE_`/`FOUNDRY_` variables, only + while a `CLAUDE_CODE_USE_*` flag for that platform is set to `1`, `true`, + `yes`, or `on` (any case) in the environment or in `settings.json` `env`; +- the provider a Codex `config.toml` selects through `model_provider` (its id, + `base_url`, `wire_api`, and `env_key`) plus the top-level `openai_base_url`; +- Claude `settings.json` credential helpers, and its `env` entries under the + same rules as the environment; +- for Kimi subscription auth, the whole Kimi `config.toml`. + +Proxy variables and a CLI's own rewrites of other config sections do not change +the route. Each time the Claude, Codex, DeepSeek, Kimi, or OpenRouter adapter +starts its CLI in a run, it recomputes the destination with the same runtime +function and fails if it no longer matches an acknowledged one. Pi and OpenCode +destinations are checked only when the run is planned. + ## Production dependency advisories CI and release validation run `pnpm security:dependency-advisories`. The gate diff --git a/docs/tutorials/first-campaign.md b/docs/tutorials/first-campaign.md index eb23371fd..3b4863e9a 100644 --- a/docs/tutorials/first-campaign.md +++ b/docs/tutorials/first-campaign.md @@ -12,9 +12,10 @@ You need: - Node.js `22.19.0` or newer and the repository-pinned `pnpm` `11.1.1` for the TypeScript workspace. - Bun `1.3` or newer for the Smithers workflow executable, plus Git. -- The CLI executables for the configured agent profiles and their credentials. - `doctor` checks executables for every configured model profile, including - profiles that are not selected for a run. +- The CLI executables for the agent profiles your topology selects, and their + credentials. `doctor` requires the executables of the agents the selected + topology and its retry fallbacks use, and lists the other configured + profiles' executables as not required. - Foundry (`forge`) and the target project's test dependencies and layout. See [Development Commands](../reference/development.md) for host runtime and diff --git a/eslint.config.js b/eslint.config.js index 615493e8d..33a2ff37b 100644 --- a/eslint.config.js +++ b/eslint.config.js @@ -9,6 +9,11 @@ const browserFiles = ["packages/dashboard/frontend/{public,src}/**/*.{js,ts,tsx} const typescriptFiles = ["packages/**/*.{ts,tsx}", "scripts/**/*.{ts,tsx}"]; const typeCheckedFiles = ["packages/*/src/**/*.ts", "packages/dashboard/frontend/src/**/*.{ts,tsx}"]; const sourceFiles = ["packages/**/*.{js,mjs,cjs,ts,tsx}", "scripts/**/*.{js,mjs,cjs,ts,tsx}"]; +// Global cyclomatic complexity ceiling (#960). It is set at the current maximum, +// with no suppressions, so no function can grow past the worst one today; lower +// it as the most complex functions are simplified. Changed lines are held to the +// stricter budgets below by `pnpm -w lint:strict:ci`. +const complexityCeilingConfigs = [{ files: sourceFiles, rules: { complexity: ["error", 90] } }]; const strictConfigs = strictLint ? [ ...tseslint.configs.strict.map((config) => ({ @@ -77,6 +82,7 @@ export default tseslint.config( }, js.configs.recommended, ...tseslint.configs.recommended, + ...complexityCeilingConfigs, ...strictConfigs, { files: ["**/*.{js,mjs,cjs,ts,tsx}"], @@ -116,6 +122,33 @@ export default tseslint.config( ] } }, + { + // Runtime templates are copied into projects and executed by Bun without a + // type check, so an undefined name would first fail when a run renders it. + // The declared globals are the placeholders the compiler substitutes. + files: ["packages/runtime/src/templates/**/*.tsx"], + languageOptions: { + globals: Object.fromEntries( + [ + "AGENT_PROMPT_TEMPLATE", + "AUTHORIZED_DEFENSIVE_SECURITY_CONTEXT", + "COMPILED_TASKS", + "DYNAMIC_GROUPS", + "MAX_DYNAMIC_NODES", + "REPLACE_PROMPT_SCHEMAS", + "RETRY_FAILURE_TEMPLATE", + "RUN_ID_LITERAL", + "RUN_ROOT_RELATIVE", + "SOURCE_PROJECT_ROOT", + "TASK_SPECS", + "UNTRUSTED_CONTENT_BOUNDARY", + "WORKFLOW_NAME", + "WORKFLOW_PATH_RELATIVE" + ].map((name) => [`__ULTRAFUZZ_${name}__`, "readonly"]) + ) + }, + rules: { "no-undef": "error" } + }, eslintConfigPrettier, ...(strictLint ? diff.configs[process.env.CI ? "flat/ci" : "flat/diff"] : []) ); diff --git a/knip.jsonc b/knip.jsonc index 09983944f..ed1a3da7a 100644 --- a/knip.jsonc +++ b/knip.jsonc @@ -2,11 +2,7 @@ "$schema": "https://unpkg.com/knip@6/schema.json", "workspaces": { ".": { - "entry": [ - "scripts/validate-audit-profile-package.mjs", - "scripts/ci/prepare-modal-benchmarks.mjs", - "scripts/ci/modal-benchmark-control-window.mjs" - ] + "entry": ["scripts/validate-audit-profile-package.mjs", "scripts/ci/prepare-modal-benchmarks.mjs"] }, "packages/dashboard": { "entry": ["frontend/public/theme-bootstrap.js"] diff --git a/packages/artifacts/src/artifact-contract-ids.ts b/packages/artifacts/src/artifact-contract-ids.ts index 67cb07fca..f01440849 100644 --- a/packages/artifacts/src/artifact-contract-ids.ts +++ b/packages/artifacts/src/artifact-contract-ids.ts @@ -62,7 +62,3 @@ export const JSON_ARTIFACT_CONTRACT_IDS = ARTIFACT_CONTRACT_IDS.filter( export function isArtifactContractId(value: unknown): value is ArtifactContractId { return typeof value === "string" && (ARTIFACT_CONTRACT_IDS as readonly string[]).includes(value); } - -export function isJsonArtifactContractId(value: unknown): value is JsonArtifactContractId { - return typeof value === "string" && (JSON_ARTIFACT_CONTRACT_IDS as readonly string[]).includes(value); -} diff --git a/packages/artifacts/src/artifact-schema-metadata.ts b/packages/artifacts/src/artifact-schema-metadata.ts index 971b55b32..3732040f9 100644 --- a/packages/artifacts/src/artifact-schema-metadata.ts +++ b/packages/artifacts/src/artifact-schema-metadata.ts @@ -64,11 +64,7 @@ export const ARTIFACT_SCHEMA_METADATA = Object.freeze({ "adminConfigBoundaryMatrixSchema", ["admin-config-surface-id-uniqueness", "admin-config-surface-joins"] ), - "agent-source-proof.schema.json": runtime("agentSourceProofJsonSchema", undefined, [ - "agent-source-proof-ref-uniqueness", - "agent-source-proof-dependency-lineage", - "agent-source-proof-commit-binding" - ]), + "agent-source-proof.schema.json": runtime("agentSourceProofJsonSchema"), "aggregation-manifest.schema.json": artifact( "ultrafuzz/aggregation-manifest@1", "aggregationManifestJsonSchema", @@ -83,8 +79,7 @@ export const ARTIFACT_SCHEMA_METADATA = Object.freeze({ ] ), "analysis-bundle.schema.json": runtime("analysisBundleManifestJsonSchema", "analysisBundleManifestSchema", [ - "analysis-bundle-path-order", - "analysis-bundle-file-digest" + "analysis-bundle-path-order" ]), "analysis-bundle-accounting-summary.schema.json": runtime( "analysisAccountingSummaryJsonSchema", @@ -116,19 +111,8 @@ export const ARTIFACT_SCHEMA_METADATA = Object.freeze({ "analysisTerminalStatusSchema", ["analysis-bundle-terminal-status-reconciliation"] ), - "artifact-manifest.schema.json": runtime("artifactManifestJsonSchema", undefined, [ - "artifact-manifest-file-path-uniqueness", - "artifact-manifest-output-path-uniqueness", - "artifact-manifest-prerequisite-node-uniqueness", - "artifact-manifest-file-digest" - ]), - "artifact-verification.schema.json": runtime("artifactVerificationJsonSchema", undefined, [ - "artifact-verification-artifact-path-uniqueness", - "artifact-verification-publication-path-uniqueness", - "artifact-verification-exactly-one-primary", - "artifact-verification-publication-digest-correspondence", - "artifact-verification-plan-contract-identity" - ]), + "artifact-manifest.schema.json": runtime("artifactManifestJsonSchema"), + "artifact-verification.schema.json": runtime("artifactVerificationJsonSchema"), "audited-differential-lanes.schema.json": artifact( "ultrafuzz/audited-differential-lanes@1", "auditedDifferentialLanesJsonSchema", @@ -147,10 +131,7 @@ export const ARTIFACT_SCHEMA_METADATA = Object.freeze({ "campaignSummarySchema", ["campaign-summary-backend-uniqueness", "campaign-summary-count-coupling"] ), - "config-redactions.schema.json": runtime("configRedactionsJsonSchema", undefined, [ - "config-redactions-path-key-equality", - "config-redactions-path-uniqueness" - ]), + "config-redactions.schema.json": runtime("configRedactionsJsonSchema"), "coverage-goal.schema.json": artifact("ultrafuzz/coverage-goal@2", "coverageGoalJsonSchema", "coverageGoalSchema", [ "coverage-goal-reconciliation" ]), @@ -241,11 +222,7 @@ export const ARTIFACT_SCHEMA_METADATA = Object.freeze({ "finding.schema.json": { role: "subschema", contractIds: [], - semanticGates: [ - "finding-campaign-provenance-coherence", - "finding-evidence-span-consistency", - "finding-projected-reference-uniqueness" - ], + semanticGates: [], typescriptExport: "findingJsonSchema", zodParser: "findingSchema" }, @@ -294,11 +271,7 @@ export const ARTIFACT_SCHEMA_METADATA = Object.freeze({ "invariant-source-proof-path-uniqueness", "invariant-source-proof-git-binding" ]), - "invariant-suite-manifest.schema.json": runtime("invariantSuiteManifestJsonSchema", undefined, [ - "invariant-suite-file-path-uniqueness", - "invariant-suite-tombstone-uniqueness", - "invariant-suite-file-tombstone-disjointness" - ]), + "invariant-suite-manifest.schema.json": runtime("invariantSuiteManifestJsonSchema"), "json-validator-preflight-success.schema.json": runtime("jsonValidatorPreflightSuccessJsonSchema", undefined, [ "json-validator-preflight-current-identity" ]), @@ -337,20 +310,7 @@ export const ARTIFACT_SCHEMA_METADATA = Object.freeze({ "lensPropertiesSchema", ["property-lens-id-uniqueness"] ), - "planned-graph.schema.json": runtime("plannedGraphJsonSchema", undefined, [ - "planned-graph-node-id-uniqueness", - "planned-graph-dependency-join", - "planned-graph-acyclicity", - "planned-graph-output-path-uniqueness", - "planned-graph-exactly-one-primary", - "planned-graph-model-fanout-uniqueness", - "planned-graph-workflow-task-uniqueness", - "planned-graph-workflow-node-join", - "planned-graph-artifact-dir-identity", - "planned-graph-loop-coupling", - "planned-graph-contract-identity", - "planned-graph-model-loop-coupling" - ]), + "planned-graph.schema.json": runtime("plannedGraphJsonSchema"), "reference-expectations.schema.json": artifact( "ultrafuzz/reference-expectations@2", "referenceExpectationsJsonSchema", @@ -383,17 +343,10 @@ export const ARTIFACT_SCHEMA_METADATA = Object.freeze({ "report-severity-classification-preservation", "report-property-provenance-join" ]), - "run-plan.schema.json": runtime("runPlanJsonSchema", undefined, ["run-plan-attempt-id-uniqueness"]), - "run-metadata.schema.json": runtime("runMetadataJsonSchema", undefined, [ - "run-metadata-workflow-id-equality", - "run-metadata-current-segment-equality", - "run-metadata-accounting-workflow-identity" - ]), - "run-state.schema.json": runtime("runStateJsonSchema", "runStateSchema", [ - "run-state-fingerprint", - "run-state-node-key-equality" - ]), - "source-run.schema.json": runtime("sourceRunJsonSchema", undefined, ["source-run-not-self"]), + "run-plan.schema.json": runtime("runPlanJsonSchema"), + "run-metadata.schema.json": runtime("runMetadataJsonSchema"), + "run-state.schema.json": runtime("runStateJsonSchema", "runStateSchema", ["run-state-node-key-equality"]), + "source-run.schema.json": runtime("sourceRunJsonSchema"), "terminal-disposition.schema.json": runtime("terminalDispositionJsonSchema", "terminalDispositionSchema"), "selected-strategies.schema.json": artifact( "ultrafuzz/selected-strategies@1", @@ -419,17 +372,7 @@ export const ARTIFACT_SCHEMA_METADATA = Object.freeze({ "severity-classification-upstream-preservation" ] ), - "smithers-task-manifest.schema.json": runtime("smithersTaskManifestJsonSchema", undefined, [ - "smithers-task-attempt-id-uniqueness", - "smithers-task-workflow-id-uniqueness", - "smithers-task-document-identity", - "smithers-task-pinned-submodule-expectation", - "smithers-task-dependency-join", - "smithers-task-dependency-acyclicity", - "smithers-task-planned-graph-coverage", - "smithers-task-planned-graph-identity", - "smithers-task-planned-graph-dependency-join" - ]), + "smithers-task-manifest.schema.json": runtime("smithersTaskManifestJsonSchema"), "strategy-detections.schema.json": artifact( "ultrafuzz/strategy-detections@1", "strategyDetectionsJsonSchema", diff --git a/packages/artifacts/src/attempt-ledger.ts b/packages/artifacts/src/attempt-ledger.ts index 94c9ad677..692d140eb 100644 --- a/packages/artifacts/src/attempt-ledger.ts +++ b/packages/artifacts/src/attempt-ledger.ts @@ -1,7 +1,4 @@ -import { isDeepStrictEqual } from "node:util"; - import { - matchesRedactedText, redactSecretsInText, redactedTextSpanCodePointLengths, SENSITIVE_REDACTION_PLACEHOLDER @@ -629,46 +626,6 @@ export function appendNodeAttempt( return appendNodeAttempts(layout, [input])[0]!; } -/** Preserve an inserted redaction only when its exact spans and all surrounding evidence replay unchanged. */ -export function reconcileNodeAttemptLedgerEntry( - prior: NodeAttemptLedgerEntry, - candidate: NodeAttemptLedgerEntry, - replayEvidence: { failureMessage?: string } = {} -): NodeAttemptLedgerEntry | undefined { - if (isDeepStrictEqual(prior, candidate)) return prior; - - const { - failure_message: priorFailureMessage, - failure_message_redaction_span_code_points: priorRedactionSpanCodePoints, - failure_message_truncated: priorFailureMessageTruncated, - ...priorWithoutFailureMessage - } = prior; - const { - failure_message: candidateFailureMessage, - failure_message_redaction_span_code_points: _candidateRedactionSpanCodePoints, - failure_message_truncated: _candidateFailureMessageTruncated, - ...candidateWithoutFailureMessage - } = candidate; - const replayFailure = - replayEvidence.failureMessage === undefined || - replayEvidence.failureMessage.includes(SENSITIVE_REDACTION_PLACEHOLDER) - ? undefined - : normalizeNodeAttemptFailureText(replayEvidence.failureMessage); - if ( - typeof priorFailureMessage === "string" && - typeof candidateFailureMessage === "string" && - priorRedactionSpanCodePoints !== undefined && - replayFailure !== undefined && - isDeepStrictEqual(priorWithoutFailureMessage, candidateWithoutFailureMessage) && - matchesRedactedText(priorFailureMessage, replayFailure, priorRedactionSpanCodePoints, { - allowObservedSuffix: priorFailureMessageTruncated === true - }) - ) { - return prior; - } - return undefined; -} - export function appendNodeAttempts( layout: Pick, inputs: readonly AppendNodeAttemptInput[] @@ -679,18 +636,17 @@ export function appendNodeAttempts( const byIdentity = new Map(existing.map((entry) => [nodeAttemptLedgerIdentity(entry), entry])); const pending: NodeAttemptLedgerEntry[] = []; const results = inputs.map((input): AppendNodeAttemptResult => { + // A Smithers occurrence is recorded once: replaying its identity returns the + // stored entry without re-deriving or comparing it. + const prior = byIdentity.get( + nodeAttemptLedgerIdentity({ + workflow_run_id: input.workflowRunId, + source_event_sequence: input.sourceEventSequence + }) + ); + if (prior !== undefined) return { entry: prior, appended: false }; const candidate = createNodeAttemptLedgerEntry(layout, input); const identity = nodeAttemptLedgerIdentity(candidate); - const prior = byIdentity.get(identity); - if (prior !== undefined) { - const reconciled = reconcileNodeAttemptLedgerEntry(prior, candidate, { - failureMessage: input.failureMessage - }); - if (reconciled === undefined) { - throw new Error(`node attempt ${identity} was already recorded with different immutable data`); - } - return { entry: reconciled, appended: false }; - } byIdentity.set(identity, candidate); pending.push(candidate); return { entry: candidate, appended: true }; diff --git a/packages/artifacts/src/events.ts b/packages/artifacts/src/events.ts index 194a22787..341f30ce4 100644 --- a/packages/artifacts/src/events.ts +++ b/packages/artifacts/src/events.ts @@ -1,7 +1,4 @@ import crypto from "node:crypto"; -import fs from "node:fs"; -import path from "node:path"; -import { isDeepStrictEqual } from "node:util"; import { redactSecretsInValue, type SecretScanMode } from "@ultrafuzz/security"; import { z } from "zod/v4"; @@ -14,19 +11,12 @@ import { canonicalUuidSchema } from "./portable-json-primitives.js"; import { type RunLayout } from "./run-layout.js"; -import { - SAFE_ID_PATTERN, - createFileDurableExclusive, - prepareSafeFilePath, - readJsonFile, - safeResolveInside, - validateSafeId -} from "./safe-paths.js"; +import { SAFE_ID_PATTERN, validateSafeId } from "./safe-paths.js"; import { schemaErrorMessage, validateWithZod, type SchemaValidationResult } from "./schema-validation.js"; import { - appendStrictJsonlRecords, + appendStrictJsonlRecordsAfterTail, + parseStrictJsonlBytes, readStrictJsonlSnapshot, - validateStrictJsonlHistory, type StrictJsonlCodec } from "./strict-jsonl.js"; import { @@ -47,7 +37,6 @@ export const DEFAULT_EVENT_REPLAY_LIMIT = 10_000; const MAX_EVENT_INDEX_FILENAME_LENGTH = 128; const EVENT_INDEX_EXTENSION = ".jsonl"; const EVENT_INDEX_DIRECT_MAX_ID_LENGTH = MAX_EVENT_INDEX_FILENAME_LENGTH - EVENT_INDEX_EXTENSION.length; -const EVENT_INDEX_LONG_DIRECTORY = "sha256"; const EVENT_INDEX_KEY_SCHEMA_VERSION = "ultrafuzz.event-index-key.v1" as const; export const EVENT_RECORD_TYPES = [ @@ -57,6 +46,7 @@ export const EVENT_RECORD_TYPES = [ "workflow-failure-unattributed", "run-recovered", "node-synced", + // No longer emitted (node-synced and node state carry the outcome); kept so existing journals replay. "node-artifacts-verified", "node-artifacts-missing", "node-controller-refinalization-intent", @@ -96,36 +86,6 @@ export interface EventQuery { limit?: number; } -export interface EventQueryFacade { - schema_version: typeof EVENT_QUERY_FACADE_SCHEMA_VERSION; - run_id: string; - append_log: string; - index_root: string; - indexes: ["run", "node", "type", "status", "timestamp"]; - filters: { - run_id: "events.index/run/.jsonl"; - node_id: "events.index/node/.jsonl"; - event_type: "events.index/type/.jsonl"; - status: "events.index/status/.jsonl"; - timestamp: "events.index/timestamp/.jsonl"; - }; - long_filters: { - run_id: "events.index/run/sha256/.jsonl"; - node_id: "events.index/node/sha256/.jsonl"; - event_type: "events.index/type/sha256/.jsonl"; - status: "events.index/status/sha256/.jsonl"; - }; - index_key_encoding: { - version: typeof EVENT_INDEX_KEY_SCHEMA_VERSION; - direct_max_id_length: number; - direct_id_path: "/.jsonl"; - long_id_path: "/sha256/.jsonl"; - digest: "sha256"; - hash_input_encoding: "utf8"; - digest_encoding: "hex"; - }; -} - const eventIdSchema = z.string().regex(/^evt-[a-f0-9]{24}$/u); const timestampSchema = canonicalTimestampSchema; const safeIdSchema = z.string().regex(SAFE_ID_PATTERN); @@ -1216,36 +1176,18 @@ export function assertEventRecord(value: unknown, recordPath = "$"): EventRecord return result.value; } -export function validateEventQueryFacade(value: unknown, recordPath = "$"): SchemaValidationResult { - return validateWithZod(eventQueryFacadeSchema as z.ZodType, value, { - path: recordPath, - code: "EVENT_QUERY_FACADE_SCHEMA_INVALID" - }); -} - -export function assertEventQueryFacade(value: unknown, recordPath = "$"): EventQueryFacade { - const result = validateEventQueryFacade(value, recordPath); - if (!result.ok || result.value === undefined) { - throw new Error(schemaErrorMessage("event query facade", result.issues)); - } - return result.value; -} - export function appendEvent(layout: RunLayout, input: AppendEventInput): EventRecord { const record = createEventRecord(layout, input); - const queryFacadePath = safeResolveInside(layout.eventsIndexDir, "query-inputs.json", "event query facade"); - assertExistingQueryFacade(layout, queryFacadePath); - const targets = [layout.eventsPath, ...eventIndexPaths(layout, record)]; - for (const target of targets) { - const codec = eventRecordCodec(layout.runId); - const existing = readStrictJsonlSnapshot(target, codec).records; - validateStrictJsonlHistory([...existing, record], codec); - } - appendEventRecord(layout.eventsPath, record, layout.root, layout.runId); - for (const target of targets.slice(1)) { - appendEventRecord(target, record, layout.root, layout.runId); - } - writeQueryFacadeInputs(layout, queryFacadePath); + // The event ID hashes the timestamp and timestamps never decrease, so only the + // trailing records that share this timestamp can repeat the ID. Checking that + // window keeps an append from costing a parse of the whole journal. + appendStrictJsonlRecordsAfterTail( + layout.eventsPath, + [record], + eventRecordCodec(layout.runId), + (existing) => existing.timestamp === record.timestamp, + layout.root + ); return record; } @@ -1279,16 +1221,6 @@ export function createEventRecord(layout: Pick, input: Appen }); } -export function appendEventRecord( - eventsPath: string, - record: EventRecord, - trustedRoot?: string, - expectedRunId?: string -): void { - const canonical = assertEventRecord(record); - appendStrictJsonlRecords(eventsPath, [canonical], eventRecordCodec(expectedRunId), trustedRoot); -} - export function replayEvents(layoutOrPath: RunLayout | string, limit = DEFAULT_EVENT_REPLAY_LIMIT): EventReplay { if (!Number.isSafeInteger(limit) || limit < 0) throw new Error("event replay limit must be a non-negative safe integer"); @@ -1302,6 +1234,11 @@ export function replayEvents(layoutOrPath: RunLayout | string, limit = DEFAULT_E }; } +/** Parse one captured event-journal byte snapshot with the rules replayEvents applies. */ +export function parseEventJournalBytes(bytes: Uint8Array, expectedRunId?: string): EventRecord[] { + return parseStrictJsonlBytes(bytes, eventRecordCodec(expectedRunId)).records; +} + export function queryEvents(layout: RunLayout, query: EventQuery = {}): EventRecord[] { const normalizedQuery = normalizeEventQuery(query); const limit = normalizedQuery.limit ?? DEFAULT_EVENT_REPLAY_LIMIT; @@ -1326,42 +1263,6 @@ export function normalizeEventQuery(query: EventQuery = {}): EventQuery { return parsed.data; } -export function readEventQueryFacade(layout: RunLayout): EventQueryFacade { - return assertEventQueryFacade(readJsonFile(path.join(layout.eventsIndexDir, "query-inputs.json"))); -} - -export function createEventQueryFacadeInputs(layout: RunLayout): EventQueryFacade { - return assertEventQueryFacade({ - schema_version: EVENT_QUERY_FACADE_SCHEMA_VERSION, - run_id: layout.runId, - append_log: path.relative(layout.root, layout.eventsPath).split(path.sep).join("/"), - index_root: path.relative(layout.root, layout.eventsIndexDir).split(path.sep).join("/"), - indexes: ["run", "node", "type", "status", "timestamp"], - filters: { - run_id: "events.index/run/.jsonl", - node_id: "events.index/node/.jsonl", - event_type: "events.index/type/.jsonl", - status: "events.index/status/.jsonl", - timestamp: "events.index/timestamp/.jsonl" - }, - long_filters: { - run_id: "events.index/run/sha256/.jsonl", - node_id: "events.index/node/sha256/.jsonl", - event_type: "events.index/type/sha256/.jsonl", - status: "events.index/status/sha256/.jsonl" - }, - index_key_encoding: { - version: EVENT_INDEX_KEY_SCHEMA_VERSION, - direct_max_id_length: EVENT_INDEX_DIRECT_MAX_ID_LENGTH, - direct_id_path: "/.jsonl", - long_id_path: "/sha256/.jsonl", - digest: "sha256", - hash_input_encoding: "utf8", - digest_encoding: "hex" - } - }); -} - export function redactValue( value: unknown, forbiddenSecretValues: readonly string[] = [], @@ -1370,29 +1271,10 @@ export function redactValue( return redactSecretsInValue(value, undefined, forbiddenSecretValues, mode); } -function eventIndexPaths(layout: RunLayout, record: EventRecord): string[] { - const nodeId = eventRecordNodeId(record); - const targets = [ - ["run", ...eventIndexPath(record.run_id)], - ["type", ...eventIndexPath(record.event_type)], - ["timestamp", ...eventIndexPath(record.timestamp.slice(0, 10))] - ]; - if (nodeId !== undefined) targets.push(["node", ...eventIndexPath(nodeId)]); - targets.push(["status", ...eventIndexPath(record.status)]); - return targets.map((segments) => prepareSafeFilePath(layout.eventsIndexDir, segments.join("/"))); -} - function eventRecordNodeId(record: EventRecord): string | undefined { return "node_id" in record ? record.node_id : undefined; } -function eventIndexPath(value: string): string[] { - const direct = `${value}${EVENT_INDEX_EXTENSION}`; - if (direct.length <= MAX_EVENT_INDEX_FILENAME_LENGTH) return [direct]; - const digest = crypto.createHash("sha256").update(value, "utf8").digest("hex"); - return [EVENT_INDEX_LONG_DIRECTORY, `${digest}${EVENT_INDEX_EXTENSION}`]; -} - function eventRecordIdentity(record: EventRecord): string { return record.event_id; } @@ -1400,6 +1282,9 @@ function eventRecordIdentity(record: EventRecord): string { function eventRecordCodec(expectedRunId?: string): StrictJsonlCodec { return { label: "event journal", + // Only the byte limit bounds this journal: a record cap, once reached, would + // fail every later sync, lifecycle and cancel call for the rest of the run. + maxRecords: Number.MAX_SAFE_INTEGER, parseRecord: (value, recordPath) => { const record = assertEventRecord(value, recordPath); if (expectedRunId !== undefined && record.run_id !== expectedRunId) { @@ -1425,16 +1310,3 @@ function eventRecordCodec(expectedRunId?: string): StrictJsonlCodec } }; } - -function assertExistingQueryFacade(layout: RunLayout, facadePath: string): void { - if (!fs.existsSync(facadePath)) return; - const actual = assertEventQueryFacade(readJsonFile(facadePath)); - const expected = createEventQueryFacadeInputs(layout); - if (!isDeepStrictEqual(actual, expected)) throw new Error("event query facade conflicts with the current run layout"); -} - -function writeQueryFacadeInputs(layout: RunLayout, facadePath: string): void { - if (fs.existsSync(facadePath)) return; - const bytes = Buffer.from(`${JSON.stringify(createEventQueryFacadeInputs(layout), null, 2)}\n`, "utf8"); - createFileDurableExclusive(facadePath, bytes, layout.root); -} diff --git a/packages/artifacts/src/finding-note-vocabulary.ts b/packages/artifacts/src/finding-note-vocabulary.ts index e8c2fc687..1073f9057 100644 --- a/packages/artifacts/src/finding-note-vocabulary.ts +++ b/packages/artifacts/src/finding-note-vocabulary.ts @@ -26,14 +26,6 @@ export const FINDING_NOTE_KEYS = [ "impact" ] as const; -/** ASCII-only, case-insensitive forms used by both the runtime and portable - * JSON-Schema grammar. Unicode compatibility folds are deliberately excluded: - * JSON Schema has no portable equivalent, and identifiers in the authority are - * ASCII exact apart from case. */ -export const FINDING_NOTE_KEYS_ASCII_CASE_INSENSITIVE_PATTERN = `(?:${FINDING_NOTE_KEYS.map((key) => - key.replace(/[A-Za-z]/gu, (character) => `[${character.toLowerCase()}${character.toUpperCase()}]`) -).join("|")})`; - export const FINDING_REPORT_ASSIGNMENT_KEY_PATTERN = "[-_0-9A-Za-z]{1,128}"; const findingReportAssignmentKey = new RegExp(`^${FINDING_REPORT_ASSIGNMENT_KEY_PATTERN}$`, "u"); @@ -129,14 +121,6 @@ export const FINDING_REPORT_MULTITERM_METADATA_KEY_PATTERNS = FINDING_REPORT_MET FINDING_REPORT_METADATA_TERM_PATTERNS.map((right) => `[-_0-9A-Za-z]*${left}[-_0-9A-Za-z]*${right}[-_0-9A-Za-z]*`) ); -/** A bounded ASCII identifier containing a report-metadata term. This defines - * a family, rather than a blacklist of guessed aliases, so producer-local - * renames such as `helper_summary` and `reachability_note` fail closed. */ -export const FINDING_REPORT_METADATA_KEY_PATTERNS = FINDING_REPORT_METADATA_TERM_GROUPS.map( - (_, index) => - `(?=${FINDING_REPORT_ASSIGNMENT_KEY_PATTERN}\\s*={1,2})[-_0-9A-Za-z]*${FINDING_REPORT_METADATA_TERM_PATTERNS[index]}[-_0-9A-Za-z]*` -); - export function isFindingReportMetadataKey(key: string): boolean { if (!findingReportAssignmentKey.test(key)) return false; const normalized = key.replace(/[A-Z]/gu, (character) => character.toLowerCase()); @@ -154,12 +138,6 @@ export function isFindingReportMetadataAliasKey(key: string): boolean { return new Set(terms).size >= 2 || (hasAliasBase && hasRenameMarker); } -export function isFindingReportEvidenceAssignmentKey(key: string): boolean { - return FINDING_REPORT_EVIDENCE_ASSIGNMENT_KEYS.includes( - key as (typeof FINDING_REPORT_EVIDENCE_ASSIGNMENT_KEYS)[number] - ); -} - export function findingReachabilityPromptVocabulary(): string { const reachabilityKey = FINDING_NOTE_KEYS[0]; return FINDING_REACHABILITY_VALUES.map((value) => `- \`${reachabilityKey}=${value}\``).join("\n"); diff --git a/packages/artifacts/src/finding-provenance.ts b/packages/artifacts/src/finding-provenance.ts deleted file mode 100644 index 483f4f8be..000000000 --- a/packages/artifacts/src/finding-provenance.ts +++ /dev/null @@ -1,506 +0,0 @@ -import fs from "node:fs"; -import path from "node:path"; - -import { FINDINGS_FILE, FINDINGS_SCHEMA_VERSION } from "./findings.js"; -import { - ArtifactPathError, - readJsonFile, - safeResolveInside, - validateNodeReference, - writeJsonDurable -} from "./safe-paths.js"; - -export interface FindingProvenance { - nodeId?: string; - producerNodeId?: string; - strategy?: string; - attemptIndex?: number; - modelId?: string; - model?: string; - modelIndex?: number; - loopIndex?: number; -} - -export interface NormalizeFindingsInput { - artifactDir: string; - relativePath?: string; - nodeId?: string; - provenance?: FindingProvenance; - preserveSourceNodes?: boolean; - requireSourceNodes?: boolean; - allowedSourceNodes?: readonly string[]; - sourceExpectations?: readonly FindingSourceExpectation[]; - requireSourceExpectation?: boolean; -} - -export interface FindingSourceExpectation { - finding_keys: readonly string[]; - source_nodes: readonly string[]; - dedupe_keys: readonly string[]; - family_ids: readonly string[]; - finding_ids: readonly string[]; - lifecycle_record?: boolean; -} - -export interface UpstreamFindingSource { - node_id: string; - artifact_path: string; - finding: unknown; -} - -export interface FindingsNormalizeReport { - schema_version: typeof FINDINGS_SCHEMA_VERSION; - source_path: string; - normalized_path: string; - count: number; - findings: Array>; -} - -export class FindingsValidationError extends Error { - constructor(message: string) { - super(message); - this.name = "FindingsValidationError"; - } -} - -/** - * Attach controller-owned provenance without repairing producer-visible finding - * fields. The caller still performs the current strict findings-contract check. - */ -export function normalizeFindings(input: NormalizeFindingsInput): FindingsNormalizeReport { - const artifactDir = path.resolve(input.artifactDir); - const relativePath = input.relativePath ?? FINDINGS_FILE; - const sourcePath = resolveFindingsSource(artifactDir, relativePath); - const raw = readJsonFile(sourcePath); - if (!Array.isArray(raw)) { - throw new FindingsValidationError(`${relativePath} must contain a findings array`); - } - const findings = raw.map((value, index) => normalizeFindingProvenance(value, index, input)); - const normalizedPath = safeResolveInside(artifactDir, relativePath, "normalized findings path"); - writeJsonDurable(normalizedPath, findings); - return { - schema_version: FINDINGS_SCHEMA_VERSION, - source_path: sourcePath, - normalized_path: normalizedPath, - count: findings.length, - findings - }; -} - -export function readFindings(artifactDir: string): Array> { - const value = readJsonFile(path.join(artifactDir, FINDINGS_FILE)); - if (!Array.isArray(value) || !value.every(isPlainRecord)) { - throw new FindingsValidationError(`${FINDINGS_FILE} must contain a findings array`); - } - return value; -} - -export function findingIdentityKeys(value: unknown): string[] { - const identity = findingIdentity(value); - return uniqueNonEmptyStrings([...identity.dedupeKeys, ...identity.familyIds, ...identity.findingIds]); -} - -/** - * Derive authoritative discovery-source expectations from verified upstream - * findings. A many-to-one lifecycle record binds its output dedupe key to the - * exact union of its authenticated source artifacts. - */ -export function buildFindingSourceExpectations(input: { - upstream: readonly UpstreamFindingSource[]; - lifecycleLedger?: unknown; - requireLifecycleCoverage?: boolean; -}): FindingSourceExpectation[] { - const upstream = input.upstream.map((entry, index) => normalizeUpstreamFindingSource(entry, index)); - const expectations: FindingSourceExpectation[] = upstream.map((entry) => ({ - finding_keys: entry.keys, - source_nodes: entry.sourceNodes, - dedupe_keys: entry.identity.dedupeKeys, - family_ids: entry.identity.familyIds, - finding_ids: entry.identity.findingIds - })); - if (input.lifecycleLedger === undefined) return expectations; - - if (!isPlainRecord(input.lifecycleLedger) || !Array.isArray(input.lifecycleLedger.records)) { - throw new FindingsValidationError("finding lifecycle ledger must contain a records array"); - } - const covered = new Set(); - const lifecycleDedupeKeys = new Set(); - for (const [recordIndex, rawRecord] of input.lifecycleLedger.records.entries()) { - if (!isPlainRecord(rawRecord)) { - throw new FindingsValidationError(`finding lifecycle record ${recordIndex} must be an object`); - } - const dedupeKey = nonEmptyString(rawRecord.dedupe_key); - if (dedupeKey === undefined) { - throw new FindingsValidationError(`finding lifecycle record ${recordIndex} requires dedupe_key`); - } - if (lifecycleDedupeKeys.has(dedupeKey)) { - throw new FindingsValidationError(`finding lifecycle ledger contains duplicate dedupe_key: ${dedupeKey}`); - } - lifecycleDedupeKeys.add(dedupeKey); - const identity = findingIdentity(rawRecord); - const keys = findingIdentityKeys(rawRecord); - if (keys.length === 0) { - throw new FindingsValidationError(`finding lifecycle record ${recordIndex} has no stable finding key`); - } - if (!Array.isArray(rawRecord.source_artifacts) || rawRecord.source_artifacts.length === 0) { - throw new FindingsValidationError(`finding lifecycle record ${recordIndex} has no source_artifacts`); - } - const sourceNodes: string[] = []; - for (const [sourceIndex, rawSource] of rawRecord.source_artifacts.entries()) { - if (!isPlainRecord(rawSource)) { - throw new FindingsValidationError( - `finding lifecycle record ${recordIndex} source_artifact ${sourceIndex} must be an object` - ); - } - const findingId = requiredLifecycleString(rawSource, "finding_id", recordIndex, sourceIndex); - const nodeId = validateNodeReference( - requiredLifecycleString(rawSource, "node_id", recordIndex, sourceIndex), - "lifecycle source artifact node ID" - ); - const matches = upstream.filter( - (candidate) => candidate.findingId === findingId && candidate.identities.includes(nodeId) - ); - if (matches.length !== 1) { - throw new FindingsValidationError( - `finding lifecycle source artifact does not identify exactly one dependency finding: ${nodeId}:${findingId}` - ); - } - const matched = matches[0]!; - if (covered.has(matched.index)) { - throw new FindingsValidationError( - `dependency finding appears more than once in the lifecycle ledger: ${nodeId}:${findingId}` - ); - } - covered.add(matched.index); - appendUnique(sourceNodes, matched.sourceNodes); - } - expectations.push({ - finding_keys: keys, - source_nodes: sourceNodes, - dedupe_keys: identity.dedupeKeys, - family_ids: identity.familyIds, - finding_ids: identity.findingIds, - lifecycle_record: true - }); - } - if (input.requireLifecycleCoverage === true && covered.size !== upstream.length) { - const missing = upstream - .filter((entry) => !covered.has(entry.index)) - .map((entry) => `${entry.nodeId}:${entry.findingId}`); - throw new FindingsValidationError(`finding lifecycle ledger omitted dependency findings: ${missing.join(", ")}`); - } - return expectations; -} - -/** Resolve a downstream finding to its exact authoritative discovery-source set. */ -export function resolveExpectedFindingSourceNodes( - finding: unknown, - expectations: readonly FindingSourceExpectation[] -): string[] | undefined { - if (!isPlainRecord(finding)) { - throw new FindingsValidationError("finding provenance candidate must be an object"); - } - return expectedSourceNodes(finding, expectations); -} - -function resolveFindingsSource(artifactDir: string, relativePath: string): string { - const findingsPath = safeResolveInside(artifactDir, relativePath, "findings source path"); - if (fs.existsSync(findingsPath) && fs.lstatSync(findingsPath).isSymbolicLink()) { - throw new ArtifactPathError("symlink-escape", `findings source path cannot be a symlink: ${relativePath}`); - } - if (!fs.existsSync(findingsPath)) throw new Error(`missing ${relativePath} in ${artifactDir}`); - return findingsPath; -} - -function normalizeFindingProvenance( - value: unknown, - index: number, - input: NormalizeFindingsInput -): Record { - if (!isPlainRecord(value)) throw new FindingsValidationError(`finding ${index} must be an object`); - const normalized = { ...value }; - const provenance = input.provenance ?? {}; - const nodeId = input.nodeId ?? provenance.nodeId; - const producerNodeId = provenance.producerNodeId ?? provenance.nodeId ?? input.nodeId; - if (producerNodeId !== undefined) { - normalized.producer_node_id = validateNodeReference(producerNodeId, "finding producer node ID"); - } - normalizeSourceNodes( - normalized, - nodeId, - producerNodeId, - input.preserveSourceNodes === true, - input.requireSourceNodes === true, - input.allowedSourceNodes, - input.sourceExpectations, - input.requireSourceExpectation === true - ); - assignIfMissing(normalized, "strategy", provenance.strategy); - assignIfMissing(normalized, "attempt_index", provenance.attemptIndex); - assignIfMissing(normalized, "model_id", provenance.modelId); - assignIfMissing(normalized, "model", provenance.model); - assignIfMissing(normalized, "model_index", provenance.modelIndex); - assignIfMissing(normalized, "loop_index", provenance.loopIndex); - return normalized; -} - -function normalizeSourceNodes( - record: Record, - nodeId: string | undefined, - producerNodeId: string | undefined, - preserveExisting: boolean, - requireSourceNodes: boolean, - allowedSourceNodes: readonly string[] | undefined, - sourceExpectations: readonly FindingSourceExpectation[] | undefined, - requireSourceExpectation: boolean -): void { - const existing = record.source_nodes; - if (existing !== undefined && !Array.isArray(existing)) { - throw new FindingsValidationError("field source_nodes must be an array of non-empty strings"); - } - const values = preserveExisting - ? [ - ...(Array.isArray(existing) ? existing : []), - typeof record.source_node_id === "string" ? record.source_node_id : undefined - ] - : [producerNodeId]; - const normalized: string[] = []; - for (const value of values) { - if (value === undefined) continue; - if (typeof value !== "string" || value.trim().length === 0) { - throw new FindingsValidationError("field source_nodes must be an array of non-empty strings"); - } - const candidate = validateNodeReference(value.trim(), "finding source node ID"); - appendUnique(normalized, [candidate]); - } - if (normalized.length === 0) { - if (requireSourceNodes) { - throw new FindingsValidationError("field source_nodes must retain at least one discovery node ID"); - } - delete record.source_nodes; - delete record.source_node_id; - delete record.producer_attempt_id; - return; - } - if (allowedSourceNodes !== undefined) { - const allowed = new Set( - allowedSourceNodes.map((sourceNode) => validateNodeReference(sourceNode, "allowed finding source node ID")) - ); - const invented = normalized.filter((sourceNode) => !allowed.has(sourceNode)); - if (invented.length > 0) { - throw new FindingsValidationError( - `field source_nodes contains IDs not present in dependency findings: ${invented.join(", ")}` - ); - } - } - const expected = expectedSourceNodes(record, sourceExpectations); - if (expected === undefined && requireSourceExpectation) { - throw new FindingsValidationError("finding does not match any dependency provenance record"); - } - if (expected !== undefined) { - if (!sameStringSet(normalized, expected)) { - throw new FindingsValidationError( - "field source_nodes does not preserve the exact dependency discovery-source union" - ); - } - normalized.splice(0, normalized.length, ...expected); - } - record.source_nodes = normalized; - record.source_node_id = normalized[0]; - if (preserveExisting) { - if (record.producer_attempt_id !== undefined) { - const attemptId = nonEmptyString(record.producer_attempt_id); - if (attemptId === undefined) { - throw new FindingsValidationError("field producer_attempt_id must be a non-empty string"); - } - record.producer_attempt_id = validateNodeReference(attemptId, "finding producer attempt ID"); - } - } else { - delete record.producer_attempt_id; - if (nodeId !== undefined && nodeId !== normalized[0]) { - record.producer_attempt_id = validateNodeReference(nodeId, "finding producer attempt ID"); - } - } -} - -function expectedSourceNodes( - record: Record, - expectations: readonly FindingSourceExpectation[] | undefined -): string[] | undefined { - if (expectations === undefined) return undefined; - const identity = findingIdentity(record); - const tiers: Array<{ - label: string; - values: readonly string[]; - expectationValues: (expectation: FindingSourceExpectation) => readonly string[]; - rejectUnknown: boolean; - }> = [ - { - label: "dedupe key", - values: identity.dedupeKeys, - expectationValues: (expectation) => expectation.dedupe_keys, - rejectUnknown: true - }, - { - label: "family ID", - values: identity.familyIds, - expectationValues: (expectation) => expectation.family_ids, - rejectUnknown: false - }, - { - label: "finding reference", - values: identity.referenceIds, - expectationValues: (expectation) => expectation.finding_ids, - rejectUnknown: true - } - ]; - let authoritative: string[] | undefined; - for (const tier of tiers) { - for (const value of tier.values) { - const matches = expectations.filter((expectation) => tier.expectationValues(expectation).includes(value)); - if (matches.length === 0) { - if (tier.rejectUnknown) { - throw new FindingsValidationError(`finding ${tier.label} does not match any dependency provenance record`); - } - continue; - } - const lifecycleMatches = matches.filter((expectation) => expectation.lifecycle_record === true); - const resolved = exactExpectedSourceNodes(lifecycleMatches.length > 0 ? lifecycleMatches : matches, tier.label); - if (authoritative === undefined) authoritative = resolved; - else if (!sameStringSet(resolved, authoritative)) { - throw new FindingsValidationError(`finding ${tier.label} conflicts with higher-priority dependency provenance`); - } - } - } - if (authoritative !== undefined || identity.ownIds.length === 0) return authoritative; - const resolvedOwnIds = identity.ownIds.flatMap((value) => { - const matches = expectations.filter((expectation) => expectation.finding_ids.includes(value)); - return matches.length === 0 ? [] : [exactExpectedSourceNodes(matches, "finding ID")]; - }); - const expected = resolvedOwnIds[0]; - if (expected === undefined) return undefined; - if (resolvedOwnIds.some((candidate) => !sameStringSet(candidate, expected))) { - throw new FindingsValidationError("finding ID values resolve to conflicting dependency provenance"); - } - return expected; -} - -function exactExpectedSourceNodes(matches: readonly FindingSourceExpectation[], identityLabel: string): string[] { - const candidates = matches.map((expectation) => - uniqueNonEmptyStrings( - expectation.source_nodes.map((sourceNode) => validateNodeReference(sourceNode, "expected finding source node ID")) - ) - ); - const expected = candidates[0] ?? []; - if (candidates.some((candidate) => !sameStringSet(candidate, expected))) { - throw new FindingsValidationError(`finding ${identityLabel} matches conflicting dependency provenance records`); - } - return expected; -} - -function normalizeUpstreamFindingSource(entry: UpstreamFindingSource, index: number) { - if (!isPlainRecord(entry.finding)) { - throw new FindingsValidationError(`dependency finding ${index} must be an object`); - } - const nodeId = validateNodeReference(entry.node_id, "dependency finding node ID"); - const findingId = nonEmptyString(entry.finding.id); - if (findingId === undefined) throw new FindingsValidationError(`finding ${index} missing required field id`); - const identity = findingIdentity(entry.finding); - const keys = findingIdentityKeys(entry.finding); - if (keys.length === 0) throw new FindingsValidationError(`dependency finding ${index} has no stable finding key`); - const sourceNodes = sourceNodesFromFinding(entry.finding, index); - const identities = [nodeId]; - if (typeof entry.finding.producer_node_id === "string") { - appendUnique(identities, [validateNodeReference(entry.finding.producer_node_id, "dependency producer node ID")]); - } - appendUnique(identities, sourceNodes); - return { index, nodeId, findingId, identities, keys, identity, sourceNodes }; -} - -interface FindingIdentity { - dedupeKeys: string[]; - familyIds: string[]; - findingIds: string[]; - referenceIds: string[]; - ownIds: string[]; -} - -function findingIdentity(value: unknown): FindingIdentity { - if (!isPlainRecord(value)) { - return { dedupeKeys: [], familyIds: [], findingIds: [], referenceIds: [], ownIds: [] }; - } - const lifecycle = isPlainRecord(value.lifecycle) ? value.lifecycle : undefined; - const referenceIds = uniqueNonEmptyStrings([value.upstream_id, value.source_finding_id, value.finding_id]); - const ownIds = uniqueNonEmptyStrings([value.id]); - return { - dedupeKeys: uniqueNonEmptyStrings([value.dedupe_key, lifecycle?.dedupe_key]), - familyIds: uniqueNonEmptyStrings([value.family_id]), - findingIds: uniqueNonEmptyStrings([...referenceIds, ...ownIds]), - referenceIds, - ownIds - }; -} - -function sourceNodesFromFinding(finding: Record, index: number): string[] { - const raw = Array.isArray(finding.source_nodes) - ? finding.source_nodes - : typeof finding.source_node_id === "string" - ? [finding.source_node_id] - : []; - if (raw.length === 0) { - throw new FindingsValidationError(`dependency finding ${index} has no discovery source nodes`); - } - const result: string[] = []; - for (const sourceNode of raw) { - const normalized = nonEmptyString(sourceNode); - if (normalized === undefined) { - throw new FindingsValidationError(`dependency finding ${index} has an invalid discovery source node`); - } - appendUnique(result, [validateNodeReference(normalized, "dependency finding source node ID")]); - } - return result; -} - -function requiredLifecycleString( - record: Record, - key: string, - recordIndex: number, - sourceIndex: number -): string { - const value = nonEmptyString(record[key]); - if (value === undefined) { - throw new FindingsValidationError( - `finding lifecycle record ${recordIndex} source_artifact ${sourceIndex} requires ${key}` - ); - } - return value; -} - -function nonEmptyString(value: unknown): string | undefined { - return typeof value === "string" && value.trim() !== "" ? value.trim() : undefined; -} - -function uniqueNonEmptyStrings(values: readonly unknown[]): string[] { - const result: string[] = []; - for (const value of values) { - const normalized = nonEmptyString(value); - if (normalized !== undefined) appendUnique(result, [normalized]); - } - return result; -} - -function appendUnique(target: string[], values: readonly string[]): void { - for (const value of values) if (!target.includes(value)) target.push(value); -} - -function sameStringSet(left: readonly string[], right: readonly string[]): boolean { - return left.length === right.length && left.every((value) => right.includes(value)); -} - -function assignIfMissing(target: Record, key: string, value: unknown): void { - if (target[key] === undefined && value !== undefined) target[key] = value; -} - -function isPlainRecord(value: unknown): value is Record { - return typeof value === "object" && value !== null && !Array.isArray(value); -} diff --git a/packages/artifacts/src/findings-schema.ts b/packages/artifacts/src/findings-schema.ts index bbae66f98..e4f45991a 100644 --- a/packages/artifacts/src/findings-schema.ts +++ b/packages/artifacts/src/findings-schema.ts @@ -1,5 +1,3 @@ -import { isDeepStrictEqual } from "node:util"; - import { z } from "zod/v4"; import { @@ -25,7 +23,7 @@ import { import { validateRegisteredJsonSchema } from "./json-schema-validator.js"; import { hasAtMostCodePoints } from "./portable-json-primitives.js"; import { NODE_REFERENCE_PATTERN } from "./safe-paths.js"; -import { schemaErrorMessage, validateWithZod, type SchemaValidationResult } from "./schema-validation.js"; +import { type SchemaValidationResult } from "./schema-validation.js"; import { jsonPointerPath } from "./lang-primitives.js"; export const FINDING_JSON_SCHEMA_ID = "urn:ultrafuzz:schema:artifacts:finding:2" as const; @@ -992,24 +990,22 @@ export const findingsJsonSchema = { } as const; export function validateFindingSchema(value: unknown, path = "$"): SchemaValidationResult { - return validateRegisteredFindingSchema(FINDING_JSON_SCHEMA_ID, findingSchema as z.ZodType, value, { + return validateRegisteredFindingSchema(FINDING_JSON_SCHEMA_ID, value, { path, code: "FINDING_SCHEMA_INVALID" }); } export function validateFindingsSchema(value: unknown, path = "$"): SchemaValidationResult { - return validateRegisteredFindingSchema( - FINDINGS_JSON_SCHEMA_ID, - findingsSchema as z.ZodType, - value, - { path, code: "FINDINGS_SCHEMA_INVALID" } - ); + return validateRegisteredFindingSchema(FINDINGS_JSON_SCHEMA_ID, value, { + path, + code: "FINDINGS_SCHEMA_INVALID" + }); } +/** The registered JSON Schema is authoritative, so its verdict is returned as is. */ function validateRegisteredFindingSchema( schemaId: string, - zodSchema: z.ZodType, value: unknown, options: { path: string; code: string } ): SchemaValidationResult { @@ -1024,35 +1020,5 @@ function validateRegisteredFindingSchema( })) }; } - - // The checked-in JSON Schema is authoritative. Zod remains only as a - // non-transforming parity assertion for typed access by existing callers. - const parity = validateWithZod(zodSchema, value, options); - if (!parity.ok) { - throw new Error( - `internal schema parity invariant violated: registered JSON Schema ${schemaId} accepted a document rejected by its retained Zod parser` - ); - } - if (!isDeepStrictEqual(parity.value, value)) { - throw new Error( - `internal schema parity invariant violated: retained Zod parser for registered JSON Schema ${schemaId} transformed its input` - ); - } return { ok: true, issues: [], value: value as T }; } - -export function assertFindingSchema(value: unknown): NormalizedFinding { - const result = validateFindingSchema(value); - if (!result.ok || !result.value) { - throw new Error(schemaErrorMessage("finding", result.issues)); - } - return result.value; -} - -export function assertFindingsSchema(value: unknown): NormalizedFinding[] { - const result = validateFindingsSchema(value); - if (!result.ok || !result.value) { - throw new Error(schemaErrorMessage("findings", result.issues)); - } - return result.value; -} diff --git a/packages/artifacts/src/findings.ts b/packages/artifacts/src/findings.ts index 5dc82579b..ca4e36a71 100644 --- a/packages/artifacts/src/findings.ts +++ b/packages/artifacts/src/findings.ts @@ -1,5 +1,4 @@ export const FINDINGS_SCHEMA_VERSION = "ultrafuzz.finding.v2" as const; -export const FINDINGS_SCHEMA_VERSIONS = [FINDINGS_SCHEMA_VERSION] as const; export const FINDINGS_FILE = "findings.json"; export const FINDING_STATUSES = [ @@ -24,12 +23,3 @@ export const TRIAGE_CLASSIFICATIONS = [ "spec-gated", "defensive-hardening" ] as const; - -export type FindingStatus = (typeof FINDING_STATUSES)[number]; -export type FindingSeverity = (typeof FINDING_SEVERITIES)[number]; -export type FindingConfidence = (typeof FINDING_CONFIDENCE_LEVELS)[number]; -export type TriageClassification = (typeof TRIAGE_CLASSIFICATIONS)[number]; - -export function isSupportedFindingsSchemaVersion(value: string): boolean { - return value === FINDINGS_SCHEMA_VERSION; -} diff --git a/packages/artifacts/src/generated-tests.ts b/packages/artifacts/src/generated-tests.ts index 70aba365d..124e1cdfa 100644 --- a/packages/artifacts/src/generated-tests.ts +++ b/packages/artifacts/src/generated-tests.ts @@ -1,39 +1,16 @@ -import fs from "node:fs"; -import path from "node:path"; - import { z } from "zod/v4"; +import { MAX_GENERATED_TEST_BUNDLE_BYTES, MAX_GENERATED_TEST_BUNDLE_ENTRIES } from "./artifact-limits.js"; import { - MAX_GENERATED_TEST_BUNDLE_BYTES, - MAX_GENERATED_TEST_BUNDLE_ENTRIES, - MAX_GENERATED_TEST_COMPANION_BYTES -} from "./artifact-limits.js"; -import { - GENERATED_TESTS_DIR, generatedTestEntriesSchema, generatedTestFrameworkSchema, generatedTestProvenanceSchema } from "./generated-test-schema.js"; import { validateRegisteredJsonSchema } from "./json-schema-validator.js"; -import { normalizeArtifactProvenance, type ArtifactProvenance } from "./manifests.js"; -import { getNodeArtifactDir, type RunLayout } from "./run-layout.js"; -import { - ArtifactPathError, - assertRegularFileInside, - ensureSafeDirectory, - normalizeSafeRelativePath, - prepareSafeFilePath, - readSinglyLinkedRegularFileSnapshotInside, - safeResolveInside, - sha256Bytes, - writeFileDurable, - writeJsonDurable -} from "./safe-paths.js"; +import { type ArtifactProvenance } from "./manifests.js"; import { validateWithZod, type SchemaValidationResult } from "./schema-validation.js"; -import { parseStrictJsonBytes } from "./strict-json.js"; export const GENERATED_TESTS_SCHEMA_VERSION = "ultrafuzz.generated-tests.v3" as const; -export const GENERATED_TESTS_MANIFEST = "generated-tests.json"; export const GENERATED_TESTS_JSON_SCHEMA_ID = "urn:ultrafuzz:schema:artifacts:generated-tests:3" as const; export { @@ -58,15 +35,6 @@ export { export type GeneratedTestProvenance = Partial>; -export interface GeneratedTestInput { - /** Exact normalized POSIX manifest path beginning with `generated-tests/`. */ - path: string; - content?: string | Uint8Array; - language?: string; - description?: string; - provenance?: GeneratedTestProvenance; -} - export interface GeneratedTestEntry { path: string; size_bytes: number; @@ -180,281 +148,6 @@ export function assertGeneratedTestBundleResourceBounds(manifest: GeneratedTestM } } -export function writeGeneratedTestManifest(input: { - layout: RunLayout; - nodeId: string; - framework: string; - tests: GeneratedTestInput[]; - supportFiles: GeneratedTestInput[]; - provenance?: GeneratedTestProvenance; -}): GeneratedTestManifest { - const provenance = normalizeArtifactProvenance(input.layout, input.nodeId, input.provenance); - preflightGeneratedTestInputShapes(input, provenance); - const nodeDir = getNodeArtifactDir(input.layout, input.nodeId); - const manifestPath = safeResolveInside(nodeDir, GENERATED_TESTS_MANIFEST, "generated tests manifest path"); - assertGeneratedTestDestinationCanBeReplaced(manifestPath); - preflightGeneratedTestFiles(nodeDir, input.tests, input.supportFiles); - getNodeArtifactDir(input.layout, input.nodeId, { create: true }); - ensureSafeDirectory(nodeDir, GENERATED_TESTS_DIR); - const generated_tests = input.tests.map((test) => writeGeneratedTestEntry(nodeDir, test, provenance)); - const support_files = input.supportFiles.map((supportFile) => - writeGeneratedTestEntry(nodeDir, supportFile, provenance) - ); - const manifest: GeneratedTestManifest = { - schema_version: GENERATED_TESTS_SCHEMA_VERSION, - run_id: input.layout.runId, - node_id: input.nodeId, - framework: input.framework, - generated_tests, - support_files, - provenance - }; - assertGeneratedTestManifestSchema(manifest); - writeJsonDurable(manifestPath, manifest); - return manifest; -} - -export function readGeneratedTestManifest(layout: RunLayout, nodeId: string): GeneratedTestManifest { - const nodeDir = getNodeArtifactDir(layout, nodeId); - const manifestPath = path.join(nodeDir, GENERATED_TESTS_MANIFEST); - return assertGeneratedTestManifestSchema( - parseStrictJsonBytes( - readSinglyLinkedRegularFileSnapshotInside(nodeDir, manifestPath, 64 * 1024 * 1024, "generated tests manifest") - ) - ); -} - -function writeGeneratedTestEntry( - nodeDir: string, - input: GeneratedTestInput, - manifestProvenance: GeneratedTestProvenance & Pick -): GeneratedTestEntry { - const safeRelativeTestPath = canonicalGeneratedTestRelativePath(input.path, "generated test path"); - const generatedTestsRoot = path.join(nodeDir, GENERATED_TESTS_DIR); - const absolutePath = prepareSafeFilePath(generatedTestsRoot, safeRelativeTestPath); - assertGeneratedTestDestinationCanBeReplaced(absolutePath); - if (input.content !== undefined) { - writeFileDurable(absolutePath, input.content); - } - assertRegularFileInside(generatedTestsRoot, absolutePath, "generated test file"); - const contents = readSinglyLinkedRegularFileSnapshotInside( - generatedTestsRoot, - absolutePath, - MAX_GENERATED_TEST_COMPANION_BYTES, - "generated test file" - ); - if (contents.length === 0) { - throw new Error(`generated test file must be non-empty: ${absolutePath}`); - } - assertStrictUtf8(contents, "generated test file", absolutePath); - return generatedTestEntryFromSnapshot(safeRelativeTestPath, input, contents, manifestProvenance); -} - -function generatedTestEntryFromSnapshot( - safeRelativeTestPath: string, - input: GeneratedTestInput, - contents: Uint8Array, - manifestProvenance: GeneratedTestProvenance & Pick -): GeneratedTestEntry { - const entry: GeneratedTestEntry = { - path: `${GENERATED_TESTS_DIR}/${safeRelativeTestPath}`, - size_bytes: contents.length, - sha256: sha256Bytes(contents), - provenance: normalizeArtifactProvenance( - { runId: manifestProvenance.run_id ?? "" }, - manifestProvenance.producer_node_id, - { - ...manifestProvenance, - ...input.provenance - } - ) - }; - if (input.language !== undefined) { - entry.language = input.language; - } - if (input.description !== undefined) { - entry.description = input.description; - } - return entry; -} - -function preflightGeneratedTestInputShapes( - input: { - layout: RunLayout; - nodeId: string; - framework: string; - tests: readonly GeneratedTestInput[]; - supportFiles: readonly GeneratedTestInput[]; - }, - provenance: GeneratedTestProvenance & Pick -): void { - const { tests, supportFiles } = input; - if (tests.length + supportFiles.length > MAX_GENERATED_TEST_BUNDLE_ENTRIES) { - throw new Error( - `generated tests manifest exceeds the ${MAX_GENERATED_TEST_BUNDLE_ENTRIES}-entry combined bundle limit` - ); - } - if (tests.length === 0 && supportFiles.length > 0) { - throw new Error("generated tests manifest cannot declare support files without a runnable generated test"); - } - const paths = new Set(); - for (const [label, entries] of [ - ["generated test", tests], - ["generated-test support", supportFiles] - ] as const) { - for (const entry of entries) { - const safeRelativePath = canonicalGeneratedTestRelativePath(entry.path, `${label} path`); - const manifestPath = `${GENERATED_TESTS_DIR}/${safeRelativePath}`; - if (paths.has(manifestPath)) { - throw new Error(`generated tests manifest repeats path ${JSON.stringify(manifestPath)}`); - } - paths.add(manifestPath); - if (entry.content !== undefined && entry.content.length === 0) { - throw new Error(`${label} file must be non-empty: ${manifestPath}`); - } - if (entry.content !== undefined && Buffer.byteLength(entry.content) > MAX_GENERATED_TEST_COMPANION_BYTES) { - throw new Error(`${label} file exceeds the ${MAX_GENERATED_TEST_COMPANION_BYTES}-byte limit: ${manifestPath}`); - } - if (entry.content !== undefined) { - assertStrictUtf8(Buffer.from(entry.content), `${label} file`, manifestPath); - } - } - } - const suppliedBytes = [...tests, ...supportFiles].reduce( - (total, entry) => total + (entry.content === undefined ? 0 : Buffer.byteLength(entry.content)), - 0 - ); - if (suppliedBytes > MAX_GENERATED_TEST_BUNDLE_BYTES) { - throw new Error( - `generated-test bundle supplied content exceeds the ${MAX_GENERATED_TEST_BUNDLE_BYTES}-byte combined limit` - ); - } - assertGeneratedTestPathsAreMaterializable(paths); - const placeholder = Buffer.from("x", "utf8"); - assertGeneratedTestManifestSchema({ - schema_version: GENERATED_TESTS_SCHEMA_VERSION, - run_id: input.layout.runId, - node_id: input.nodeId, - framework: input.framework, - generated_tests: tests.map((entry) => - generatedTestEntryFromSnapshot( - canonicalGeneratedTestRelativePath(entry.path, "generated test path"), - entry, - placeholder, - provenance - ) - ), - support_files: supportFiles.map((entry) => - generatedTestEntryFromSnapshot( - canonicalGeneratedTestRelativePath(entry.path, "generated-test support path"), - entry, - placeholder, - provenance - ) - ), - provenance - }); -} - -function preflightGeneratedTestFiles( - nodeDir: string, - tests: readonly GeneratedTestInput[], - supportFiles: readonly GeneratedTestInput[] -): void { - const generatedTestsRoot = path.join(nodeDir, GENERATED_TESTS_DIR); - const existingFiles: Array<{ label: string; manifestPath: string; absolutePath: string; sizeBytes: number }> = []; - let totalBytes = 0; - for (const [label, entries] of [ - ["generated test", tests], - ["generated-test support", supportFiles] - ] as const) { - for (const entry of entries) { - const safeRelativePath = canonicalGeneratedTestRelativePath(entry.path, `${label} path`); - const manifestPath = `${GENERATED_TESTS_DIR}/${safeRelativePath}`; - const absolutePath = safeResolveInside(generatedTestsRoot, safeRelativePath, `${label} path`); - assertGeneratedTestDestinationCanBeReplaced(absolutePath); - if (entry.content !== undefined) { - totalBytes += Buffer.byteLength(entry.content); - continue; - } - assertRegularFileInside(generatedTestsRoot, absolutePath, `${label} file`); - const sizeBytes = fs.lstatSync(absolutePath).size; - totalBytes += sizeBytes; - existingFiles.push({ label, manifestPath, absolutePath, sizeBytes }); - } - } - if (totalBytes > MAX_GENERATED_TEST_BUNDLE_BYTES) { - throw new Error( - `generated-test bundle exceeds the ${MAX_GENERATED_TEST_BUNDLE_BYTES}-byte combined companion limit` - ); - } - for (const { label, manifestPath, sizeBytes } of existingFiles) { - if (sizeBytes === 0) { - throw new Error(`${label} file must be non-empty: ${manifestPath}`); - } - if (sizeBytes > MAX_GENERATED_TEST_COMPANION_BYTES) { - throw new Error(`${label} file exceeds the ${MAX_GENERATED_TEST_COMPANION_BYTES}-byte limit: ${manifestPath}`); - } - } - for (const { label, manifestPath, absolutePath } of existingFiles) { - const contents = readSinglyLinkedRegularFileSnapshotInside( - generatedTestsRoot, - absolutePath, - MAX_GENERATED_TEST_COMPANION_BYTES, - `${label} file` - ); - if (contents.length === 0) { - throw new Error(`${label} file must be non-empty: ${manifestPath}`); - } - assertStrictUtf8(contents, `${label} file`, manifestPath); - } -} - -function assertStrictUtf8(contents: Uint8Array, label: string, filePath: string): void { - try { - new TextDecoder("utf-8", { fatal: true }).decode(contents); - } catch { - throw new Error(`${label} must be strict UTF-8 text: ${filePath}`); - } -} - -function assertGeneratedTestDestinationCanBeReplaced(filePath: string): void { - let fileStats: fs.Stats; - try { - fileStats = fs.lstatSync(filePath); - } catch (error) { - if ((error as NodeJS.ErrnoException).code === "ENOENT") { - return; - } - throw error; - } - if (fileStats.isSymbolicLink()) { - throw new ArtifactPathError("symlink-escape", `generated-test bundle destination cannot be a symlink: ${filePath}`); - } - if (!fileStats.isFile()) { - throw new ArtifactPathError("not-file", `generated-test bundle destination must be a regular file: ${filePath}`); - } - if (fileStats.nlink !== 1) { - throw new ArtifactPathError("hard-link", `generated-test bundle destination must be singly linked: ${filePath}`); - } -} - -function canonicalGeneratedTestRelativePath(value: string, label: string): string { - const prefix = `${GENERATED_TESTS_DIR}/`; - if (!value.startsWith(prefix)) { - throw new ArtifactPathError("noncanonical-path", `${label} must begin with ${JSON.stringify(prefix)}`); - } - const relativePath = value.slice(prefix.length); - const normalized = normalizeSafeRelativePath(relativePath, label); - if (relativePath !== normalized) { - throw new ArtifactPathError( - "noncanonical-path", - `${label} must already be a normalized relative POSIX path: ${JSON.stringify(value)}` - ); - } - return normalized; -} - function assertGeneratedTestPathsAreMaterializable(paths: ReadonlySet): void { const directoryOrderedPaths = [...paths].sort((left, right) => { const leftDirectory = `${left}/`; diff --git a/packages/artifacts/src/goal-plan.ts b/packages/artifacts/src/goal-plan.ts index b45e679f2..69577db23 100644 --- a/packages/artifacts/src/goal-plan.ts +++ b/packages/artifacts/src/goal-plan.ts @@ -123,6 +123,12 @@ const replacementValue = nonEmptyString message: "Replacement values must be a human-readable title, not a serialized JSON record: keep the record in the " + "artifact directory and reference it by path so the goal sentence stays one sentence" + }) + // The dynamic-node renderer resolves `{{...}}` inside replacement values, and a reference it cannot + // bind throws inside the workflow render, which fails the whole run on every resume. A label has no + // reason to carry template syntax, so reject it here, where the failure is goal-plan's own verify. + .refine((value) => !value.includes("{{"), { + message: "Replacement values must be plain-text labels and must not contain template braces '{{'" }); const replacementsSchema = z @@ -332,7 +338,6 @@ const applicabilityDecisionSchema = z }); export const GOAL_LANE_KINDS = ["threat", "class", "roaming"] as const; -export type GoalLaneKind = (typeof GOAL_LANE_KINDS)[number]; /** * One goal lane: a named unit of hunting work and the concrete node IDs it owns. @@ -598,9 +603,6 @@ export const goalPlanJsonSchema = { } as Record; export type GoalPlan = z.infer; -export type ThreatGoalPlanItem = z.infer; -export type ClassGoalPlanItem = z.infer; -export type ApplicabilityDecision = z.infer; export function validateGoalPlan(value: unknown, path = "$"): SchemaValidationResult { return validateWithZod(goalPlanSchema, value, { path, code: "GOAL_PLAN_SCHEMA_INVALID" }); diff --git a/packages/artifacts/src/index.ts b/packages/artifacts/src/index.ts index ed7e2fc5b..7836d9e2e 100644 --- a/packages/artifacts/src/index.ts +++ b/packages/artifacts/src/index.ts @@ -10,7 +10,6 @@ export * from "./artifact-contracts.js"; export * from "./artifact-schema-metadata.js"; export * from "./cloud-selected-task.js"; export * from "./findings.js"; -export * from "./finding-provenance.js"; export * from "./findings-schema.js"; export * from "./finding-note-vocabulary.js"; export * from "./coverage-evidence.js"; diff --git a/packages/artifacts/src/invariant-ledger.ts b/packages/artifacts/src/invariant-ledger.ts index 24b98698a..c26c2fe1c 100644 --- a/packages/artifacts/src/invariant-ledger.ts +++ b/packages/artifacts/src/invariant-ledger.ts @@ -1,6 +1,6 @@ import { z } from "zod/v4"; -import { validateWithZod, type SchemaValidationIssue, type SchemaValidationResult } from "./schema-validation.js"; +import { validateWithZod, type SchemaValidationResult } from "./schema-validation.js"; import { executeSemanticGates } from "./semantic-gates.js"; export const INVARIANT_LEDGER_SCHEMA_VERSION = "ultrafuzz.invariant-evidence-ledger.v1" as const; @@ -71,8 +71,8 @@ export const invariantLedgerSchema = z schema_version: z.literal(INVARIANT_LEDGER_SCHEMA_VERSION), entries: z.array(invariantLedgerEntrySchema), inventory_rows: z.array(invariantInventoryRowSchema).optional(), - // Present only on a ledger that records no invariant at all. The gate requires it there - // (issue #292): an empty ledger is otherwise indistinguishable from an agent that did not + // Present only on a ledger that records no invariant at all. The refinement below requires it + // there (issue #292): an empty ledger is otherwise indistinguishable from an agent that did not // look, and nothing reads `scan_probes[].result`, so probe text alone cannot carry that claim. no_invariants_justification: nonEmptyString.optional(), scan_probes: z @@ -142,7 +142,6 @@ export const invariantLedgerSchema = z }); export type InvariantLedgerEntry = z.infer; -export type InvariantInventoryRow = z.infer; export type InvariantLedgerArtifact = z.infer; export function validateInvariantLedgerSchema( @@ -203,10 +202,6 @@ function publicInvariantLedgerSemanticPath(semanticPath: string, message: string : semanticPath; } -export function invariantLedgerSchemaIssues(value: unknown, path = "$"): SchemaValidationIssue[] { - return validateInvariantLedgerSchema(value, path).issues; -} - export const invariantLedgerJsonSchema = { $schema: "https://json-schema.org/draft/2020-12/schema", $id: "urn:ultrafuzz:schema:artifacts:invariant-evidence-ledger:1", diff --git a/packages/artifacts/src/invariant-source-proof.ts b/packages/artifacts/src/invariant-source-proof.ts index 06607eaca..e089b7038 100644 --- a/packages/artifacts/src/invariant-source-proof.ts +++ b/packages/artifacts/src/invariant-source-proof.ts @@ -1,6 +1,6 @@ import { z } from "zod/v4"; -import { validateWithZod, type SchemaValidationIssue, type SchemaValidationResult } from "./schema-validation.js"; +import { validateWithZod, type SchemaValidationResult } from "./schema-validation.js"; import { executeSemanticGate } from "./semantic-gates.js"; export const INVARIANT_SOURCE_PROOF_SCHEMA_VERSION = "ultrafuzz.invariant-source-proof.v1" as const; @@ -41,7 +41,6 @@ export const invariantSourceProofSchema = z.strictObject({ }); export type InvariantSourceProof = z.infer; -export type InvariantSourceProofFile = z.infer; export function validateInvariantSourceProofSchema( value: unknown, @@ -68,10 +67,6 @@ function prefixedSemanticPath(rootPath: string, semanticPath: string): string { return semanticPath === "$" ? rootPath : `${rootPath}${semanticPath.slice(1)}`; } -export function invariantSourceProofSchemaIssues(value: unknown, path = "$"): SchemaValidationIssue[] { - return validateInvariantSourceProofSchema(value, path).issues; -} - export const invariantSourceProofJsonSchema = { $schema: "https://json-schema.org/draft/2020-12/schema", $id: "urn:ultrafuzz:schema:artifacts:invariant-source-proof:1", diff --git a/packages/artifacts/src/json-validator-preflight.ts b/packages/artifacts/src/json-validator-preflight.ts index 2464ad75a..64036fbcf 100644 --- a/packages/artifacts/src/json-validator-preflight.ts +++ b/packages/artifacts/src/json-validator-preflight.ts @@ -108,7 +108,6 @@ export interface JsonValidatorPreflightExpectedIdentity { schemaId: string; schemaSha256: string; schemaBundleSha256: string; - validatorBuild: string; artifactSha256: string; } @@ -134,7 +133,6 @@ export function parseJsonValidatorPreflightSuccessEnvelope( schemaId: binding!.schema_id, schemaSha256: binding!.schema_sha256, schemaBundleSha256: binding!.schema_bundle_sha256, - validatorBuild: binding!.validator_build, artifactSha256: ARTIFACT_VALIDATOR_SMOKE_FIXTURE_SHA256 }; const gates = executeSchemaSemanticGates(JSON_VALIDATOR_PREFLIGHT_SUCCESS_SCHEMA_FILENAME, { @@ -144,7 +142,6 @@ export function parseJsonValidatorPreflightSuccessEnvelope( schemaId: expected.schemaId, schemaSha256: expected.schemaSha256, schemaBundleSha256: expected.schemaBundleSha256, - validatorBuild: expected.validatorBuild, artifactSha256: expected.artifactSha256 } } diff --git a/packages/artifacts/src/manifests.ts b/packages/artifacts/src/manifests.ts index 022b6a0ff..7a50d1965 100644 --- a/packages/artifacts/src/manifests.ts +++ b/packages/artifacts/src/manifests.ts @@ -8,7 +8,6 @@ import { listSafeFiles, normalizeSafeRelativePath, prepareSafeFilePath, - safeResolveInside, sha256File, validateNodeReference, validateSafeId, @@ -443,49 +442,6 @@ export function readArtifactManifest(layout: RunLayout, nodeId: string): Artifac return manifest as ArtifactManifest; } -export interface RunArtifactIndexEntry extends ArtifactManifestEntry { - node_id: string; -} - -export interface RunArtifactIndex { - schema_version: string; - run_id: string; - artifacts: RunArtifactIndexEntry[]; -} - -export function buildRunArtifactIndex(layout: RunLayout): RunArtifactIndex { - const artifacts: RunArtifactIndexEntry[] = []; - if (!fs.existsSync(layout.artifactsDir)) { - return { schema_version: ARTIFACT_MANIFEST_SCHEMA_VERSION, run_id: layout.runId, artifacts }; - } - - for (const dirent of fs.readdirSync(layout.artifactsDir, { withFileTypes: true })) { - if (!dirent.isDirectory()) { - continue; - } - const nodeId = validateSafeId(dirent.name, "node ID"); - const nodeDir = path.join(layout.artifactsDir, nodeId); - const manifestPath = path.join(nodeDir, ARTIFACT_MANIFEST_FILE); - if (!fs.existsSync(manifestPath)) { - continue; - } - assertRegularFileInside(layout.artifactsDir, manifestPath, "artifact manifest path"); - const manifest = readArtifactManifest(layout, nodeId); - for (const file of manifest.files) { - const artifactPath = safeResolveInside(nodeDir, file.path, "artifact manifest file path"); - assertRegularFileInside(nodeDir, artifactPath, "artifact manifest file path"); - artifacts.push({ - ...file, - node_id: nodeId, - path: `artifacts/${nodeId}/${file.path}` - }); - } - } - - artifacts.sort((left, right) => left.node_id.localeCompare(right.node_id) || left.path.localeCompare(right.path)); - return { schema_version: ARTIFACT_MANIFEST_SCHEMA_VERSION, run_id: layout.runId, artifacts }; -} - function assertValidArtifactManifest(value: unknown): void { const validation = validateArtifactManifest(value); if (validation.ok) return; diff --git a/packages/artifacts/src/planned-graph.ts b/packages/artifacts/src/planned-graph.ts index ebbaf72dd..3506866d3 100644 --- a/packages/artifacts/src/planned-graph.ts +++ b/packages/artifacts/src/planned-graph.ts @@ -5,7 +5,6 @@ import { } from "./artifact-contract-ids.js"; import { CANONICAL_ARTIFACT_RELATIVE_PATH_PATTERN } from "./artifact-path-primitives.js"; import { MAX_RETRY_CHAIN_ATTEMPTS } from "./artifact-limits.js"; -import { artifactContractDefinition, artifactContractSchemaBinding } from "./artifact-contracts.js"; import { validateRegisteredJsonSchema, type JsonSchemaValidationResult } from "./json-schema-validator.js"; import { readRegularFileSnapshot } from "./schema-registry.js"; import { parseStrictJsonBytes } from "./strict-json.js"; @@ -386,22 +385,11 @@ export function assertPlannedGraph(value: unknown): PlannedGraphDocument { /** * Parse a graph whose exact bytes are already authenticated by workflow-control - * evidence. A controller-only upgrade may change the digest of the complete - * schema bundle without changing this graph's contract schema. Keep every - * contract-specific binding strict and admit only that historical bundle ID. + * evidence. Sealed and freshly planned graphs are checked identically: shape and + * internal consistency, never the reading build's contract registry. */ export function assertSealedPlannedGraph(value: unknown): PlannedGraphDocument { - const shape = validatePlannedGraph(value); - if (!shape.ok) { - throw new Error( - `planned graph is schema-invalid: ${shape.issues - .map((issue) => `${issue.instancePath || "/"} ${issue.message}`) - .join("; ")}` - ); - } - const graph = value as PlannedGraphDocument; - assertPlannedGraphSemantics(graph, { allowHistoricalSchemaBundle: true }); - return graph; + return assertPlannedGraph(value); } export function readPlannedGraphDocument(filePath: string): PlannedGraphDocument { @@ -415,90 +403,13 @@ export function readPlannedGraphDocument(filePath: string): PlannedGraphDocument ); } -export function assertPlannedGraphSemantics( - graph: PlannedGraphDocument, - options: { allowHistoricalSchemaBundle?: boolean } = {} -): void { +export function assertPlannedGraphSemantics(graph: PlannedGraphDocument): void { const nodes = new Map(); const workflowTaskIds = new Set(); for (const node of graph.nodes) { if (nodes.has(node.id)) throw new Error(`planned graph repeats node ID ${JSON.stringify(node.id)}`); nodes.set(node.id, node); - const artifactIdentity = node.dynamic_generated?.storage_id ?? node.id; - const expectedArtifactDirs = node.model_fanout.map((model) => `artifacts/${model.attempt_id ?? artifactIdentity}`); - const expectedPrimaryArtifactDir = - node.dynamic_generated === undefined - ? `artifacts/${node.id}` - : (expectedArtifactDirs[0] ?? `artifacts/${artifactIdentity}`); - if (node.artifact_dir !== expectedPrimaryArtifactDir) { - throw new Error(`planned graph artifact_dir does not match node ID ${JSON.stringify(node.id)}`); - } - if ( - node.dynamic_generated !== undefined && - JSON.stringify(node.artifact_dirs ?? []) !== JSON.stringify(expectedArtifactDirs) - ) { - throw new Error(`planned graph dynamic artifact directories do not match its generated attempts`); - } - if (node.loop.index >= node.loop.count || node.loop.attempt_index !== node.loop.index) { - throw new Error(`planned graph node ${JSON.stringify(node.id)} has inconsistent loop coordinates`); - } - - const outputPaths = new Set(); - let primaryCount = 0; - for (const output of node.outputs) { - if (outputPaths.has(output.path)) { - throw new Error( - `planned graph node ${JSON.stringify(node.id)} repeats output path ${JSON.stringify(output.path)}` - ); - } - outputPaths.add(output.path); - if (output.primary) primaryCount += 1; - const definition = artifactContractDefinition(output.contract); - if (output.contract_digest !== definition.digest) { - throw new Error(`planned graph output contract digest changed for ${JSON.stringify(output.path)}`); - } - const binding = artifactContractSchemaBinding(output.contract); - if ( - (binding === undefined && output.schema_file !== undefined) || - (binding !== undefined && - (output.schema_file !== binding.schema_file || - output.schema_id !== binding.schema_id || - output.schema_sha256 !== binding.schema_sha256 || - (options.allowHistoricalSchemaBundle !== true && - output.schema_bundle_sha256 !== binding.schema_bundle_sha256) || - output.validator_build !== binding.validator_build)) - ) { - throw new Error(`planned graph output schema binding changed for ${JSON.stringify(output.path)}`); - } - } - if (primaryCount !== 1) { - throw new Error(`planned graph node ${JSON.stringify(node.id)} must identify exactly one primary output`); - } - - const modelKeys = new Set(); - const modelAttemptIds = new Set(); - for (const model of node.model_fanout) { - const key = `${model.model_profile_id}\u0000${model.model_index}\u0000${model.loop_index}\u0000${model.attempt_index}`; - if (modelKeys.has(key)) { - throw new Error(`planned graph node ${JSON.stringify(node.id)} repeats a model-fanout identity`); - } - modelKeys.add(key); - if (model.loop_index !== node.loop.index) { - throw new Error(`planned graph node ${JSON.stringify(node.id)} has a model bound to another loop`); - } - const expectedAttemptId = - node.model_fanout.length <= 1 - ? artifactIdentity - : `${artifactIdentity}__model_${model.model_index}__attempt_${model.attempt_index}`; - if (model.attempt_id !== undefined && model.attempt_id !== expectedAttemptId) { - throw new Error(`planned graph node ${JSON.stringify(node.id)} has an inconsistent model attempt ID`); - } - const attemptId = model.attempt_id ?? expectedAttemptId; - if (modelAttemptIds.has(attemptId)) { - throw new Error(`planned graph node ${JSON.stringify(node.id)} repeats a model attempt ID`); - } - modelAttemptIds.add(attemptId); - } + assertPlannedNodeSemantics(node); for (const taskId of node.workflow?.task_node_ids ?? []) { if (workflowTaskIds.has(taskId)) throw new Error(`planned graph repeats workflow task ID ${JSON.stringify(taskId)}`); @@ -532,3 +443,74 @@ export function assertPlannedGraphSemantics( }; for (const nodeId of nodes.keys()) visit(nodeId); } + +function assertPlannedNodeSemantics(node: PlannedGraphNodeDocument): void { + const artifactIdentity = node.dynamic_generated?.storage_id ?? node.id; + const expectedArtifactDirs = node.model_fanout.map((model) => `artifacts/${model.attempt_id ?? artifactIdentity}`); + const expectedPrimaryArtifactDir = + node.dynamic_generated === undefined + ? `artifacts/${node.id}` + : (expectedArtifactDirs[0] ?? `artifacts/${artifactIdentity}`); + if (node.artifact_dir !== expectedPrimaryArtifactDir) { + throw new Error(`planned graph artifact_dir does not match node ID ${JSON.stringify(node.id)}`); + } + if ( + node.dynamic_generated !== undefined && + JSON.stringify(node.artifact_dirs ?? []) !== JSON.stringify(expectedArtifactDirs) + ) { + throw new Error(`planned graph dynamic artifact directories do not match its generated attempts`); + } + if (node.loop.index >= node.loop.count || node.loop.attempt_index !== node.loop.index) { + throw new Error(`planned graph node ${JSON.stringify(node.id)} has inconsistent loop coordinates`); + } + assertPlannedNodeOutputs(node); + assertPlannedNodeModelFanout(node, artifactIdentity); +} + +function assertPlannedNodeOutputs(node: PlannedGraphNodeDocument): void { + const outputPaths = new Set(); + let primaryCount = 0; + for (const output of node.outputs) { + if (outputPaths.has(output.path)) { + throw new Error( + `planned graph node ${JSON.stringify(node.id)} repeats output path ${JSON.stringify(output.path)}` + ); + } + outputPaths.add(output.path); + if (output.primary) primaryCount += 1; + // An output's contract digest and schema binding record the build that planned it, and the + // planned-graph schema already requires a binding exactly for schema-backed contracts. They + // are not re-derived against the reading build: that stranded in-flight runs whenever a + // rebuild changed any of them (#921). Artifact gates validate against the recorded schema. + } + if (primaryCount !== 1) { + throw new Error(`planned graph node ${JSON.stringify(node.id)} must identify exactly one primary output`); + } +} + +function assertPlannedNodeModelFanout(node: PlannedGraphNodeDocument, artifactIdentity: string): void { + const modelKeys = new Set(); + const modelAttemptIds = new Set(); + for (const model of node.model_fanout) { + const key = [model.model_profile_id, model.model_index, model.loop_index, model.attempt_index].join("\u0000"); + if (modelKeys.has(key)) { + throw new Error(`planned graph node ${JSON.stringify(node.id)} repeats a model-fanout identity`); + } + modelKeys.add(key); + if (model.loop_index !== node.loop.index) { + throw new Error(`planned graph node ${JSON.stringify(node.id)} has a model bound to another loop`); + } + const expectedAttemptId = + node.model_fanout.length <= 1 + ? artifactIdentity + : `${artifactIdentity}__model_${String(model.model_index)}__attempt_${String(model.attempt_index)}`; + if (model.attempt_id !== undefined && model.attempt_id !== expectedAttemptId) { + throw new Error(`planned graph node ${JSON.stringify(node.id)} has an inconsistent model attempt ID`); + } + const attemptId = model.attempt_id ?? expectedAttemptId; + if (modelAttemptIds.has(attemptId)) { + throw new Error(`planned graph node ${JSON.stringify(node.id)} repeats a model attempt ID`); + } + modelAttemptIds.add(attemptId); + } +} diff --git a/packages/artifacts/src/property-provenance.ts b/packages/artifacts/src/property-provenance.ts index 25aaf7954..fd9e1edbb 100644 --- a/packages/artifacts/src/property-provenance.ts +++ b/packages/artifacts/src/property-provenance.ts @@ -3,12 +3,7 @@ import { z } from "zod/v4"; import { canonicalArtifactRelativePathSchema } from "./artifact-path-primitives.js"; import { validateRegisteredJsonSchema } from "./json-schema-validator.js"; import { canonicalTimestampSchema, hasAtMostCodePoints } from "./portable-json-primitives.js"; -import { - schemaErrorMessage, - validateWithZod, - type SchemaValidationIssue, - type SchemaValidationResult -} from "./schema-validation.js"; +import { type SchemaValidationIssue, type SchemaValidationResult } from "./schema-validation.js"; import { jsonPointerPath } from "./lang-primitives.js"; export const PROPERTIES_SCHEMA_VERSION = "ultrafuzz.properties.v2" as const; @@ -275,19 +270,13 @@ export interface PropertyCampaignArtifact { failures: PropertyCampaignFailure[]; } -export type FindingFuzzerBackendProvenance = +type FindingFuzzerBackendProvenance = | { present: false; valid: true; backends: readonly [] } | { present: true; valid: false; backends: readonly [] } | { present: true; valid: true; backends: readonly string[] }; -/** - * Read backend provenance owned by a deduplicated campaign finding. Keeping - * this parser beside the campaign artifact contract gives the runtime gate and - * final-report verification one interpretation of the singular/plural fields. - */ -export function findingFuzzerBackendProvenance( - finding: Readonly> -): FindingFuzzerBackendProvenance { +/** Read backend provenance owned by a deduplicated campaign finding: its singular or plural field. */ +function findingFuzzerBackendProvenance(finding: Readonly>): FindingFuzzerBackendProvenance { const hasBackend = Object.prototype.hasOwnProperty.call(finding, "fuzzer_backend"); const hasBackends = Object.prototype.hasOwnProperty.call(finding, "fuzzer_backends"); if (!hasBackend && !hasBackends) { @@ -1045,20 +1034,15 @@ export function validateLensPropertiesSchema( value: unknown, path = "$" ): SchemaValidationResult { - return validateRegisteredPropertySchema( - PROPERTY_LENS_JSON_SCHEMA_ID, - lensPropertiesSchema as z.ZodType, - value, - { - path, - code: "PROPERTY_LENS_SCHEMA_INVALID" - } - ); + return validateRegisteredPropertySchema(PROPERTY_LENS_JSON_SCHEMA_ID, value, { + path, + code: "PROPERTY_LENS_SCHEMA_INVALID" + }); } +/** The registered JSON Schema is authoritative, so its verdict is returned as is. */ function validateRegisteredPropertySchema( schemaId: string, - zodSchema: z.ZodType, value: unknown, options: { path: string; code: string } ): SchemaValidationResult { @@ -1073,15 +1057,6 @@ function validateRegisteredPropertySchema( })) }; } - - // The checked-in JSON Schema is authoritative. Zod remains only as a - // non-transforming parity assertion for typed access by existing callers. - const parity = validateWithZod(zodSchema, value, options); - if (!parity.ok) { - throw new Error( - `internal schema parity invariant violated: registered JSON Schema ${schemaId} accepted a document rejected by its retained Zod parser` - ); - } return { ok: true, issues: [], value: value as T }; } @@ -1089,27 +1064,17 @@ export function validateReferenceExpectationsSchema( value: unknown, path = "$" ): SchemaValidationResult { - return validateRegisteredPropertySchema( - REFERENCE_EXPECTATIONS_JSON_SCHEMA_ID, - referenceExpectationsSchema as z.ZodType, - value, - { - path, - code: "REFERENCE_EXPECTATIONS_SCHEMA_INVALID" - } - ); + return validateRegisteredPropertySchema(REFERENCE_EXPECTATIONS_JSON_SCHEMA_ID, value, { + path, + code: "REFERENCE_EXPECTATIONS_SCHEMA_INVALID" + }); } export function validatePropertiesSchema(value: unknown, path = "$"): SchemaValidationResult { - return validateRegisteredPropertySchema( - PROPERTIES_JSON_SCHEMA_ID, - propertiesSchema as z.ZodType, - value, - { - path, - code: "PROPERTIES_SCHEMA_INVALID" - } - ); + return validateRegisteredPropertySchema(PROPERTIES_JSON_SCHEMA_ID, value, { + path, + code: "PROPERTIES_SCHEMA_INVALID" + }); } export function validateImplementedPropertiesSchema( @@ -1117,9 +1082,8 @@ export function validateImplementedPropertiesSchema( path = "$", options: { requireSelection?: boolean } = {} ): SchemaValidationResult { - const result = validateRegisteredPropertySchema( + const result = validateRegisteredPropertySchema( IMPLEMENTED_PROPERTIES_JSON_SCHEMA_ID, - implementedPropertiesSchema as z.ZodType, value, { path, @@ -1145,23 +1109,10 @@ export function validatePropertyCampaignSchema( value: unknown, path = "$" ): SchemaValidationResult { - return validateRegisteredPropertySchema( - PROPERTY_CAMPAIGN_JSON_SCHEMA_ID, - propertyCampaignSchema as z.ZodType, - value, - { - path, - code: "PROPERTY_CAMPAIGN_SCHEMA_INVALID" - } - ); -} - -export function assertPropertiesSchema(value: unknown): PropertiesArtifact { - const result = validatePropertiesSchema(value); - if (!result.ok || result.value === undefined) { - throw new Error(schemaErrorMessage("properties", result.issues)); - } - return result.value; + return validateRegisteredPropertySchema(PROPERTY_CAMPAIGN_JSON_SCHEMA_ID, value, { + path, + code: "PROPERTY_CAMPAIGN_SCHEMA_INVALID" + }); } export function validatePropertyReferences( diff --git a/packages/artifacts/src/report-observation.ts b/packages/artifacts/src/report-observation.ts index ae2a11413..e086ed327 100644 --- a/packages/artifacts/src/report-observation.ts +++ b/packages/artifacts/src/report-observation.ts @@ -48,6 +48,5 @@ export const reportObservedCompletionSchema = z.strictObject({ }); export type ReportVerification = z.infer; -export type ReportVerificationReasonCode = ReportVerification["reason_codes"][number]; export type ReportObservedCompletion = z.infer; export type ObservedReportCompletion = ReportObservedCompletion; diff --git a/packages/artifacts/src/run-layout.ts b/packages/artifacts/src/run-layout.ts index 48c59ba56..6ce66e84c 100644 --- a/packages/artifacts/src/run-layout.ts +++ b/packages/artifacts/src/run-layout.ts @@ -2,7 +2,6 @@ import crypto from "node:crypto"; import fs from "node:fs"; import path from "node:path"; -import { createEventQueryFacadeInputs } from "./events.js"; import { assertPlannedGraph, PLANNED_GRAPH_SCHEMA_VERSION, type PlannedGraphDocument } from "./planned-graph.js"; import { CONFIG_REDACTIONS_SCHEMA_VERSION, @@ -44,7 +43,6 @@ export interface RunLayout { eventsPath: string; usageLedgerPath: string; attemptLedgerPath: string; - eventsIndexDir: string; } export interface CreateRunLayoutInput { @@ -80,13 +78,7 @@ export function createRunLayout(input: CreateRunLayoutInput): RunLayout { assertNoSymlinkComponents(guardRoot, root, "run root"); const layout = layoutForRunRoot(root, runId); - for (const directory of [ - layout.root, - layout.artifactsDir, - layout.workspacesDir, - layout.eventsIndexDir, - layout.reviewDir - ]) { + for (const directory of [layout.root, layout.artifactsDir, layout.workspacesDir, layout.reviewDir]) { fs.mkdirSync(directory, { recursive: true }); } @@ -165,11 +157,6 @@ export function createRunLayout(input: CreateRunLayoutInput): RunLayout { if ((input.overwrite ?? false) || !fs.existsSync(layout.attemptLedgerPath)) { writeFileDurable(layout.attemptLedgerPath, ""); } - writeJsonIfNeeded( - path.join(layout.eventsIndexDir, "query-inputs.json"), - createEventQueryFacadeInputs(layout), - input.overwrite ?? false - ); return layout; } @@ -193,8 +180,7 @@ export function layoutForRunRoot(root: string, runId = path.basename(root)): Run statePath: path.join(absoluteRoot, "state.json"), eventsPath: path.join(absoluteRoot, "events.jsonl"), usageLedgerPath: path.join(absoluteRoot, "usage.jsonl"), - attemptLedgerPath: path.join(absoluteRoot, "attempts.jsonl"), - eventsIndexDir: path.join(absoluteRoot, "events.index") + attemptLedgerPath: path.join(absoluteRoot, "attempts.jsonl") }; } diff --git a/packages/artifacts/src/runtime-schemas.ts b/packages/artifacts/src/runtime-schemas.ts index 93bbd4972..8892d3c0d 100644 --- a/packages/artifacts/src/runtime-schemas.ts +++ b/packages/artifacts/src/runtime-schemas.ts @@ -192,29 +192,6 @@ export const agentSourceProofJsonSchema = { } } as const; -export interface AgentSourceProof { - schema_version: typeof AGENT_SOURCE_PROOF_SCHEMA_VERSION; - attempt_id: string; - commit: string; - tree: string; - base_ref: "refs/heads/ultrafuzz-pinned"; - refs: Array<{ name: string; object: string }>; - remotes: []; - revision_count: 1; - commit_object_count: 1; - dependencies: { - schema_version: "ultrafuzz.pinned-submodules-expectation.v1"; - source_commit: string; - source_tree: string; - manifest_sha256: string; - top_level_roots: string[]; - recursive_gitlinks: Array<{ path: string; commit: string; tree: string }>; - entry_count: number; - file_count: number; - total_file_bytes: number; - } | null; -} - export interface ArtifactVerificationEntry { path: string; contract: ArtifactContractId; diff --git a/packages/artifacts/src/safe-paths.ts b/packages/artifacts/src/safe-paths.ts index 395fe012e..8698f7d27 100644 --- a/packages/artifacts/src/safe-paths.ts +++ b/packages/artifacts/src/safe-paths.ts @@ -368,40 +368,6 @@ export function writeJsonDurable(filePath: string, value: unknown): void { writeFileDurable(filePath, `${JSON.stringify(value, null, 2)}\n`); } -export function appendLineDurable(filePath: string, line: string, trustedRoot?: string): void { - const directory = path.dirname(filePath); - if (trustedRoot !== undefined) { - assertNoSymlinkComponents(trustedRoot, directory, "append directory"); - } - fs.mkdirSync(directory, { recursive: true }); - if (trustedRoot !== undefined) { - assertNoSymlinkComponents(trustedRoot, filePath, "append path"); - } - const fd = fs.openSync( - filePath, - fs.constants.O_APPEND | fs.constants.O_CREAT | fs.constants.O_WRONLY | fs.constants.O_NOFOLLOW, - 0o600 - ); - runWithClosedDescriptor(fd, `failed to durably append ${filePath} and close its descriptor`, () => { - if (!fs.fstatSync(fd).isFile()) { - throw new ArtifactPathError("not-file", `append path must be a regular file: ${filePath}`); - } - if (trustedRoot !== undefined) { - assertNoSymlinkComponents(trustedRoot, filePath, "append path"); - } - const bytes = Buffer.from(line.endsWith("\n") ? line : `${line}\n`, "utf8"); - const written = fs.writeSync(fd, bytes); - if (written !== bytes.length) { - throw new ArtifactPathError( - "short-write", - `durable append wrote ${written} of ${bytes.length} bytes: ${filePath}` - ); - } - fs.fsyncSync(fd); - }); - fsyncDirectory(directory); -} - /** * Durably creates a new regular file without accepting an intervening writer. * @@ -482,43 +448,6 @@ export function appendBytesDurableAt( fsyncDirectory(path.dirname(filePath)); } -/** - * Durably discards an unterminated trailing fragment. - * - * The expected size fences the repair decision against a concurrent append; - * hard links are rejected so truncation cannot mutate another named file. - */ -export function truncateDurable( - filePath: string, - length: number, - options: { expectedSize: number; trustedRoot?: string } -): void { - if (options.trustedRoot !== undefined) { - assertNoSymlinkComponents(options.trustedRoot, filePath, "truncate path"); - } - const fd = fs.openSync(filePath, fs.constants.O_WRONLY | fs.constants.O_NOFOLLOW | fs.constants.O_NONBLOCK); - try { - const stat = fs.fstatSync(fd); - if (!stat.isFile()) { - throw new ArtifactPathError("not-file", `truncate path must be a regular file: ${filePath}`); - } - if (stat.nlink !== 1) { - throw new ArtifactPathError("not-file", `truncate path must not be hard-linked: ${filePath}`); - } - if (stat.size !== options.expectedSize) { - throw new ArtifactPathError("not-file", `truncate path changed size before truncation: ${filePath}`); - } - if (length > stat.size) { - throw new ArtifactPathError("not-file", `truncate length exceeds the file size: ${filePath}`); - } - fs.ftruncateSync(fd, length); - fs.fsyncSync(fd); - } finally { - fs.closeSync(fd); - } - fsyncDirectory(path.dirname(filePath)); -} - export function readJsonFile(filePath: string): T { return parseStrictJsonBytes(readRegularFileSnapshot(filePath, 64 * 1024 * 1024), { maxBytes: 64 * 1024 * 1024, diff --git a/packages/artifacts/src/sealed-schema-registry.ts b/packages/artifacts/src/sealed-schema-registry.ts index 7b7925ea4..82c5a84da 100644 --- a/packages/artifacts/src/sealed-schema-registry.ts +++ b/packages/artifacts/src/sealed-schema-registry.ts @@ -4,6 +4,7 @@ import path from "node:path"; import { artifactSchemaRegistry, + DEFAULT_MAX_JSON_INSTANCE_BYTES, readRegularFileSnapshot, type ArtifactSchemaRegistryEntry } from "./schema-registry.js"; @@ -15,7 +16,13 @@ const MAX_REGISTERED_PATTERNS = 256; const MAX_REGISTERED_BUNDLE_PATTERNS = 1_024; const MAX_REGISTERED_PATTERN_LENGTH = 1_024; -/** Load a complete, physical schema bundle from an authenticated execution snapshot. */ +/** + * Load the complete physical schema bundle sealed in an execution snapshot, exactly as it was sealed. + * A later build may add, remove or re-version schema files, so the bundle is not compared with the + * installed registry; callers compare its bundle digest with the one an artifact was planned against. + * A sealed bundle is only used to validate, so a file carries the installed build's contract, gate + * and export metadata only when that build still registers it under the same `$id`. + */ export function artifactSchemaRegistryFromDirectory(directory: string): readonly ArtifactSchemaRegistryEntry[] { const resolved = path.resolve(directory); const lexical = fs.lstatSync(resolved); @@ -28,19 +35,11 @@ export function artifactSchemaRegistryFromDirectory(directory: string): readonly .readdirSync(resolved) .filter((filename) => filename.endsWith(".schema.json")) .sort(); - const unknown = filenames.filter((filename) => !currentByFilename.has(filename)); - const missing = [...currentByFilename.keys()].filter((filename) => !filenames.includes(filename)); - if (unknown.length > 0 || missing.length > 0) { - throw new Error( - `schema registry mismatch${unknown.length > 0 ? `; unregistered: ${unknown.join(", ")}` : ""}${missing.length > 0 ? `; missing: ${missing.join(", ")}` : ""}` - ); - } let bundleBytes = 0; let bundlePatterns = 0; return Object.freeze( filenames.map((filename): ArtifactSchemaRegistryEntry => { - const current = currentByFilename.get(filename)!; const snapshot = readRegularFileSnapshot(path.join(resolved, filename), MAX_REGISTERED_SCHEMA_BYTES); bundleBytes += snapshot.byteLength; if (bundleBytes > MAX_REGISTERED_BUNDLE_BYTES) { @@ -56,7 +55,10 @@ export function artifactSchemaRegistryFromDirectory(directory: string): readonly if (parsed.$schema !== "https://json-schema.org/draft/2020-12/schema") { throw new Error(`schema must declare Draft 2020-12: ${filename}`); } - if (parsed.$id !== current.id) throw new Error(`sealed schema $id changed for ${filename}`); + const id = parsed.$id; + if (typeof id !== "string" || id.length === 0 || id.includes("#")) { + throw new Error(`schema must have a fragment-free non-empty $id: ${filename}`); + } bundlePatterns += assertRegisteredPatternLimits(parsed, filename); if (bundlePatterns > MAX_REGISTERED_BUNDLE_PATTERNS) { throw new Error(`registered schema bundle exceeds the ${MAX_REGISTERED_BUNDLE_PATTERNS}-pattern limit`); @@ -65,8 +67,19 @@ export function artifactSchemaRegistryFromDirectory(directory: string): readonly for (const reference of localReferences) { if (/^https?:/iu.test(reference)) throw new Error(`remote schema reference is forbidden: ${reference}`); } + const current = currentByFilename.get(filename); return Object.freeze({ - ...current, + ...(current?.id === id + ? current + : { + filename, + id, + role: "subschema" as const, + contractIds: Object.freeze([]), + maxInstanceBytes: DEFAULT_MAX_JSON_INSTANCE_BYTES, + semanticGates: Object.freeze([]), + typescriptExport: "" + }), sha256: sha256(snapshot), schema: deepFreezeJson(parsed), localReferences: Object.freeze(localReferences) diff --git a/packages/artifacts/src/semantic-gates.ts b/packages/artifacts/src/semantic-gates.ts index c0bcc42a1..561ea50f8 100644 --- a/packages/artifacts/src/semantic-gates.ts +++ b/packages/artifacts/src/semantic-gates.ts @@ -3,7 +3,6 @@ import fs from "node:fs"; import path from "node:path"; import { isDeepStrictEqual } from "node:util"; -import { artifactContractDefinition, artifactContractSchemaBinding } from "./artifact-contracts.js"; import { ARTIFACT_SCHEMA_METADATA, type ArtifactSchemaFilename } from "./artifact-schema-metadata.js"; import { artifactMetadataCompletenessIssues, metadataOmission } from "./artifact-validation.js"; import { findingNoteAssignmentIssue } from "./findings-schema.js"; @@ -38,7 +37,6 @@ export interface SemanticFilesystemContext { export interface SemanticGitContext { commit: string; tree: string; - refs?: Readonly>; baseCommit?: string; baseTree?: string; resultTree?: string; @@ -164,22 +162,12 @@ export interface SemanticArtifactSetContext { differentialArtifacts?: SemanticDifferentialArtifactsContext; } -export interface SemanticPlannedGraphContext { - node?: unknown; - document?: unknown; -} - export interface SemanticAttemptLedgerContext { entries: readonly unknown[]; /** Trusted entries from a source run that may be referenced by reuse evidence. */ sourceEntries?: readonly unknown[]; } -export interface SemanticRuntimeStateContext { - graphFingerprint: string; - configFingerprint: string; -} - export interface SemanticArtifactIdentityContext { runId: string; nodeId: string; @@ -228,7 +216,6 @@ export interface SemanticValidatorPreflightContext { schemaId: string; schemaSha256: string; schemaBundleSha256: string; - validatorBuild: string; artifactSha256: string; } @@ -287,10 +274,8 @@ export interface SemanticGateContext { filesystem?: SemanticFilesystemContext; git?: SemanticGitContext; artifactSet?: SemanticArtifactSetContext; - plannedGraph?: SemanticPlannedGraphContext; artifactIdentity?: SemanticArtifactIdentityContext; attemptLedger?: SemanticAttemptLedgerContext; - runtimeState?: SemanticRuntimeStateContext; usageLedger?: SemanticUsageLedgerContext; eventLog?: SemanticEventLogContext; validatorPreflight?: SemanticValidatorPreflightContext; @@ -429,7 +414,8 @@ function jsonValidatorPreflightIdentityIssues(document: unknown, context: Semant ["$.data.schema.id", at(document, ["data", "schema", "id"]), expected.schemaId], ["$.data.schema.sha256", at(document, ["data", "schema", "sha256"]), expected.schemaSha256], ["$.data.schema.bundle_sha256", at(document, ["data", "schema", "bundle_sha256"]), expected.schemaBundleSha256], - ["$.data.schema.validator_build", at(document, ["data", "schema", "validator_build"]), expected.validatorBuild], + // `validator_build` is provenance: the validator that reported the identity may come from another + // build of the same schemas, which must not stop an in-flight run's preflight (#921). ["$.data.artifact_sha256", at(document, ["data", "artifact_sha256"]), expected.artifactSha256] ] as const; return checks.flatMap(([pathValue, actual, wanted]) => @@ -3046,34 +3032,6 @@ function appendBigintEqualityIssue( } } -function artifactVerificationDigestIssues(document: unknown): SemanticGateIssue[] { - const publications = new Map(); - for (const row of arrayAt(document, ["publications"])) { - const rowPath = stringField(row, "path"); - const digest = stringField(row, "sha256"); - if (rowPath !== undefined && digest !== undefined) publications.set(rowPath, digest); - } - const issues: SemanticGateIssue[] = []; - for (const [index, artifact] of arrayAt(document, ["artifacts"]).entries()) { - const artifactPath = stringField(artifact, "path"); - const digest = stringField(artifact, "sha256"); - if (artifactPath !== undefined && publications.get(artifactPath) !== digest) { - issues.push( - issue( - `$.artifacts[${index}].sha256`, - `Publication digest does not correspond to artifact ${JSON.stringify(artifactPath)}` - ) - ); - } - } - return issues; -} - -function exactlyOnePrimaryIssues(document: unknown, key: string): SemanticGateIssue[] { - const count = arrayAt(document, [key]).filter((row) => booleanField(row, "primary") === true).length; - return count === 1 ? [] : [issue(`$.${key}`, `Exactly one ${key} entry must be primary; found ${count}`)]; -} - function attemptOrderIssues(document: unknown): SemanticGateIssue[] { const lifecycle = at(document, ["lifecycle"]); const started = stringField(lifecycle, "started_at"); @@ -4889,74 +4847,6 @@ function externalizedStateJoinIssues(document: unknown): SemanticGateIssue[] { return issues; } -function findingProjectedReferenceIssues(document: unknown): SemanticGateIssue[] { - if (!isRecord(document)) return []; - const groups: Array<{ - items: readonly unknown[]; - path: string; - project: (row: Readonly>) => string | undefined; - label: string; - }> = [ - { - items: arrayAt(document, ["family_variants"]), - path: "$.family_variants", - project: (row) => stringField(row, "id"), - label: "family variant ID" - }, - { - items: arrayAt(document, ["family_variants"]), - path: "$.family_variants", - project: (row) => stringField(row, "dedupe_key"), - label: "family variant dedupe key" - }, - { - items: arrayAt(document, ["related_findings"]), - path: "$.related_findings", - project: (row) => stringField(row, "id"), - label: "related finding ID" - }, - { - items: arrayAt(document, ["lifecycle", "source_artifacts"]), - path: "$.lifecycle.source_artifacts", - project: (row) => { - const values = [row.path, row.node_id, row.finding_id]; - return values.some((value) => value === undefined) ? undefined : JSON.stringify(values); - }, - label: "lifecycle source reference" - }, - { - items: arrayAt(document, ["lifecycle", "strategy_hits"]), - path: "$.lifecycle.strategy_hits", - project: (row) => - JSON.stringify([ - row.strategy, - row.attempt_index ?? null, - row.model_id ?? null, - row.model_index ?? null, - row.loop_index ?? null - ]), - label: "strategy hit identity" - } - ]; - const issues = projectedUniquenessIssues(groups); - const contributions = arrayAt(document, ["contributing_backend_failures"]); - const seen = new Set(); - for (const [index, contribution] of contributions.entries()) { - const key = - typeof contribution === "string" - ? JSON.stringify([null, contribution]) - : isRecord(contribution) - ? JSON.stringify([contribution.fuzzer_backend, contribution.failure_id]) - : undefined; - if (key === undefined) continue; - if (seen.has(key)) { - issues.push(issue(`$.contributing_backend_failures[${index}]`, `Duplicate contributing backend failure ${key}`)); - } - seen.add(key); - } - return issues; -} - function findingEvidenceSpanIssues(document: unknown, findingPath = "$"): SemanticGateIssue[] { if (!isRecord(document)) return []; const issues = evidenceArraySpanIssues(arrayAt(document, ["evidence"]), `${findingPath}.evidence`); @@ -5094,208 +4984,6 @@ function invariantLedgerJoinIssues(document: unknown): SemanticGateIssue[] { return issues; } -function plannedNodes(document: unknown): readonly unknown[] { - return arrayAt(document, ["nodes"]); -} - -function plannedNodeIdIssues(document: unknown): SemanticGateIssue[] { - return uniqueFieldGate([["nodes"]], "id", "planned graph node ID")(document, {}); -} - -function plannedDependencyJoinIssues(document: unknown): SemanticGateIssue[] { - const ids = new Set(plannedNodes(document).flatMap((node) => stringField(node, "id") ?? "")); - ids.delete(""); - const issues: SemanticGateIssue[] = []; - for (const [nodeIndex, node] of plannedNodes(document).entries()) { - const nodeId = stringField(node, "id"); - for (const [dependencyIndex, dependency] of stringArray(at(node, ["depends_on"])).entries()) { - if (!ids.has(dependency)) { - issues.push( - issue( - `$.nodes[${nodeIndex}].depends_on[${dependencyIndex}]`, - `Unknown planned dependency ${JSON.stringify(dependency)}` - ) - ); - } else if (dependency === nodeId) { - issues.push( - issue(`$.nodes[${nodeIndex}].depends_on[${dependencyIndex}]`, "A planned node cannot depend on itself") - ); - } - } - } - return issues; -} - -function plannedAcyclicityIssues(document: unknown): SemanticGateIssue[] { - const nodes = new Map(); - for (const node of plannedNodes(document)) { - const id = stringField(node, "id"); - if (id !== undefined) nodes.set(id, node); - } - const visiting = new Set(); - const visited = new Set(); - let cycle: string | undefined; - const visit = (nodeId: string): void => { - if (cycle !== undefined || visited.has(nodeId)) return; - if (visiting.has(nodeId)) { - cycle = nodeId; - return; - } - visiting.add(nodeId); - for (const dependency of stringArray(at(nodes.get(nodeId), ["depends_on"]))) { - if (nodes.has(dependency)) visit(dependency); - } - visiting.delete(nodeId); - visited.add(nodeId); - }; - for (const nodeId of nodes.keys()) visit(nodeId); - return cycle === undefined - ? [] - : [issue("$.nodes", `Planned graph contains a dependency cycle at ${JSON.stringify(cycle)}`)]; -} - -function plannedOutputPathIssues(document: unknown): SemanticGateIssue[] { - return plannedNodes(document).flatMap((node, nodeIndex) => - projectedUniquenessIssues([ - { - items: arrayAt(node, ["outputs"]), - path: `$.nodes[${nodeIndex}].outputs`, - project: (row) => stringField(row, "path"), - label: "planned output path" - } - ]) - ); -} - -function plannedPrimaryIssues(document: unknown): SemanticGateIssue[] { - return plannedNodes(document).flatMap((node, nodeIndex) => { - const count = arrayAt(node, ["outputs"]).filter((output) => booleanField(output, "primary") === true).length; - return count === 1 - ? [] - : [ - issue( - `$.nodes[${nodeIndex}].outputs`, - `Planned node must identify exactly one primary output; found ${count}` - ) - ]; - }); -} - -function plannedModelFanoutIssues(document: unknown): SemanticGateIssue[] { - return plannedNodes(document).flatMap((node, nodeIndex) => - projectedUniquenessIssues([ - { - items: arrayAt(node, ["model_fanout"]), - path: `$.nodes[${nodeIndex}].model_fanout`, - project: (row) => JSON.stringify([row.model_profile_id, row.model_index, row.loop_index, row.attempt_index]), - label: "model-fanout identity" - } - ]) - ); -} - -function plannedWorkflowTaskIssues(document: unknown): SemanticGateIssue[] { - const seen = new Set(); - const issues: SemanticGateIssue[] = []; - for (const [nodeIndex, node] of plannedNodes(document).entries()) { - for (const [taskIndex, taskId] of stringArray(at(node, ["workflow", "task_node_ids"])).entries()) { - if (seen.has(taskId)) { - issues.push( - issue( - `$.nodes[${nodeIndex}].workflow.task_node_ids[${taskIndex}]`, - `Duplicate workflow task ID ${JSON.stringify(taskId)}` - ) - ); - } - seen.add(taskId); - } - } - return issues; -} - -function plannedWorkflowJoinIssues(document: unknown): SemanticGateIssue[] { - return plannedNodes(document).flatMap((node, nodeIndex) => { - const workflow = at(node, ["workflow"]); - if (!isRecord(workflow)) return []; - const nodeId = stringField(workflow, "node_id"); - return nodeId !== undefined && !stringArray(workflow.task_node_ids).includes(nodeId) - ? [issue(`$.nodes[${nodeIndex}].workflow.node_id`, "Workflow node_id must be present in task_node_ids")] - : []; - }); -} - -function plannedArtifactDirIssues(document: unknown): SemanticGateIssue[] { - return plannedNodes(document).flatMap((node, nodeIndex) => { - const id = stringField(node, "id"); - const artifactDir = stringField(node, "artifact_dir"); - return id !== undefined && artifactDir !== `artifacts/${id}` - ? [issue(`$.nodes[${nodeIndex}].artifact_dir`, "artifact_dir must be derived from the planned node ID")] - : []; - }); -} - -function plannedLoopIssues(document: unknown): SemanticGateIssue[] { - return plannedNodes(document).flatMap((node, nodeIndex) => { - const loop = at(node, ["loop"]); - const index = numberField(loop, "index"); - const count = numberField(loop, "count"); - const attempt = numberField(loop, "attempt_index"); - return index !== undefined && count !== undefined && (index >= count || attempt !== index) - ? [issue(`$.nodes[${nodeIndex}].loop`, "Planned loop coordinates are inconsistent")] - : []; - }); -} - -function plannedContractIdentityIssues(document: unknown): SemanticGateIssue[] { - const issues: SemanticGateIssue[] = []; - for (const [nodeIndex, node] of plannedNodes(document).entries()) { - for (const [outputIndex, output] of arrayAt(node, ["outputs"]).entries()) { - const contract = stringField(output, "contract"); - if (contract === undefined) continue; - let definition: ReturnType; - try { - definition = artifactContractDefinition(contract as Parameters[0]); - } catch { - continue; - } - if (stringField(output, "contract_digest") !== definition.digest) { - issues.push( - issue( - `$.nodes[${nodeIndex}].outputs[${outputIndex}].contract_digest`, - "Planned output contract digest changed" - ) - ); - } - const binding = artifactContractSchemaBinding(contract as Parameters[0]); - const bindingFields = [ - "schema_file", - "schema_id", - "schema_sha256", - "schema_bundle_sha256", - "validator_build" - ] as const; - if ( - (binding === undefined && bindingFields.some((field) => isRecord(output) && output[field] !== undefined)) || - (binding !== undefined && bindingFields.some((field) => isRecord(output) && output[field] !== binding[field])) - ) { - issues.push(issue(`$.nodes[${nodeIndex}].outputs[${outputIndex}]`, "Planned output schema binding changed")); - } - } - } - return issues; -} - -function plannedModelLoopIssues(document: unknown): SemanticGateIssue[] { - return plannedNodes(document).flatMap((node, nodeIndex) => { - const loopIndex = numberField(at(node, ["loop"]), "index"); - return arrayAt(node, ["model_fanout"]).flatMap((model, modelIndex) => - numberField(model, "loop_index") === loopIndex - ? [] - : [issue(`$.nodes[${nodeIndex}].model_fanout[${modelIndex}].loop_index`, "Model is bound to another loop")] - ); - }); -} - function propertySourceProjectedIssues(document: unknown): SemanticGateIssue[] { const seen = new Set(); const issues: SemanticGateIssue[] = []; @@ -5345,404 +5033,6 @@ function runStateNodeKeyIssues(document: unknown): SemanticGateIssue[] { ); } -function smithersTasks(document: unknown): readonly unknown[] { - return arrayAt(document, ["tasks"]); -} - -function smithersWorkflowIdentityIssues(document: unknown): SemanticGateIssue[] { - return [ - ...uniqueFieldGate([["tasks"]], "smithersNodeId", "Smithers workflow node ID")(document, {}), - ...uniqueFieldGate([["tasks"]], "verifierSmithersNodeId", "Smithers verifier node ID")(document, {}) - ]; -} - -function sameUnknownArray(left: unknown, right: unknown): boolean { - return Array.isArray(left) && Array.isArray(right) && JSON.stringify(left) === JSON.stringify(right); -} - -function smithersDocumentIdentityIssues(document: unknown): SemanticGateIssue[] { - const runId = stringField(document, "run_id"); - const workflowName = stringField(document, "workflow_name"); - const issues: SemanticGateIssue[] = []; - for (const [index, task] of smithersTasks(document).entries()) { - if (!isRecord(task)) continue; - const taskPath = `$.tasks[${index}]`; - const attemptId = stringField(task, "attemptId"); - if (attemptId !== undefined && stringField(task, "smithersNodeId") !== `node:${attemptId}`) { - issues.push(issue(`${taskPath}.smithersNodeId`, "Smithers workflow node ID must be derived from attemptId")); - } - if (attemptId !== undefined && stringField(task, "verifierSmithersNodeId") !== `verify:${attemptId}`) { - issues.push( - issue(`${taskPath}.verifierSmithersNodeId`, "Smithers verifier node ID must be derived from attemptId") - ); - } - const metadata = at(task, ["metadata"]); - if (isRecord(metadata)) { - const metadataRun = at(metadata, ["run"]); - if ( - stringField(metadataRun, "ultrafuzzRunId") !== runId || - stringField(metadataRun, "smithersWorkflowName") !== workflowName - ) { - issues.push(issue(`${taskPath}.metadata.run`, "Smithers task run metadata does not match its document")); - } - const metadataNode = at(metadata, ["node"]); - if ( - stringField(metadataNode, "attemptId") !== attemptId || - stringField(metadataNode, "concreteNodeId") !== stringField(task, "concreteNodeId") || - stringField(metadataNode, "logicalNodeId") !== stringField(task, "logicalNodeId") - ) { - issues.push(issue(`${taskPath}.metadata.node`, "Smithers task node metadata does not match its envelope")); - } - const metadataModel = at(metadata, ["model"]); - if ( - stringField(metadataModel, "agentRef") !== stringField(task, "agentRef") || - (isRecord(metadataModel) ? metadataModel.modelName : undefined) !== task.modelName || - (isRecord(metadataModel) ? metadataModel.reasoningEffort : undefined) !== task.reasoningEffort - ) { - issues.push(issue(`${taskPath}.metadata.model`, "Smithers task model metadata does not match its envelope")); - } - if (!sameUnknownArray(task.dependencies, at(metadata, ["dependencies", "attemptIds"]))) { - issues.push( - issue(`${taskPath}.metadata.dependencies.attemptIds`, "Dependency attempt metadata does not match") - ); - } - if (!sameUnknownArray(task.dependencySmithersNodeIds, at(metadata, ["dependencies", "smithersNodeIds"]))) { - issues.push( - issue(`${taskPath}.metadata.dependencies.smithersNodeIds`, "Dependency workflow metadata does not match") - ); - } - const timeout = at(metadata, ["timeout"]); - const retryPolicy = at(metadata, ["retryPolicy"]); - const timeoutMs = numberField(task, "timeoutMs"); - const retries = numberField(task, "retries"); - if ( - numberField(timeout, "milliseconds") !== timeoutMs || - numberField(timeout, "heartbeatTimeoutMs") !== numberField(task, "heartbeatTimeoutMs") || - numberField(retryPolicy, "smithersRetries") !== retries || - (retries !== undefined && numberField(retryPolicy, "maxAttempts") !== retries + 1) || - (timeoutMs !== undefined && numberField(timeout, "seconds") !== Math.ceil(timeoutMs / 1_000)) - ) { - issues.push(issue(`${taskPath}.metadata.timeout`, "Smithers timeout or retry metadata does not match")); - } - const execution = at(task, ["execution"]); - const metadataExecution = at(metadata, ["execution"]); - if ( - stringField(execution, "mode") !== stringField(metadataExecution, "mode") || - (isRecord(execution) ? execution.provider : undefined) !== - (isRecord(metadataExecution) ? metadataExecution.provider : undefined) || - JSON.stringify(at(execution, ["resources"])) !== JSON.stringify(at(metadataExecution, ["resources"])) || - stringField(at(metadata, ["artifacts"]), "dir") !== stringField(task, "artifactDir") - ) { - issues.push(issue(`${taskPath}.metadata.execution`, "Smithers execution or artifact metadata does not match")); - } - } - } - return issues; -} - -function smithersPinnedSubmoduleIssues(document: unknown): SemanticGateIssue[] { - const expectation = at(document, ["pinned_submodules"]); - if (expectation === null || !isRecord(expectation)) return []; - const issues: SemanticGateIssue[] = []; - const roots = stringArray(at(expectation, ["top_level_roots"])); - const gitlinks = arrayAt(expectation, ["recursive_gitlinks"]); - const gitlinkPaths = gitlinks.flatMap((entry) => stringField(entry, "path") ?? []); - issues.push(...pinnedSubmodulePortablePathIssues(expectation, "$.pinned_submodules")); - const canonical = (values: readonly string[]): boolean => - new Set(values).size === values.length && values.every((value, index) => index === 0 || values[index - 1]! < value); - if (!canonical(roots)) { - issues.push(issue("$.pinned_submodules.top_level_roots", "Pinned submodule roots must be unique and ordered")); - } - if (!canonical(gitlinkPaths)) { - issues.push( - issue("$.pinned_submodules.recursive_gitlinks", "Pinned submodule gitlinks must be unique and path-ordered") - ); - } - for (const [index, root] of roots.entries()) { - if (!gitlinkPaths.includes(root)) { - issues.push(issue(`$.pinned_submodules.top_level_roots[${index}]`, "Pinned submodule root is not a gitlink")); - } - if (roots.some((candidate, candidateIndex) => candidateIndex !== index && root.startsWith(`${candidate}/`))) { - issues.push(issue(`$.pinned_submodules.top_level_roots[${index}]`, "Pinned submodule roots overlap")); - } - } - const entryCount = numberField(expectation, "entry_count"); - const fileCount = numberField(expectation, "file_count"); - if (entryCount !== undefined && fileCount !== undefined && fileCount > entryCount) { - issues.push(issue("$.pinned_submodules.file_count", "Pinned submodule file count exceeds entry count")); - } - return issues; -} - -function smithersDependencyJoinIssues(document: unknown): SemanticGateIssue[] { - const tasks = smithersTasks(document); - const byAttempt = new Map( - tasks.flatMap((task) => { - const id = stringField(task, "attemptId"); - return id === undefined ? [] : [[id, task] as const]; - }) - ); - const byVerifier = new Map( - tasks.flatMap((task) => { - const id = stringField(task, "verifierSmithersNodeId"); - return id === undefined ? [] : [[id, task] as const]; - }) - ); - const issues: SemanticGateIssue[] = []; - for (const [taskIndex, task] of tasks.entries()) { - const attemptId = stringField(task, "attemptId"); - const dependencies = stringArray(at(task, ["dependencies"])); - const joined = new Set(); - for (const [dependencyIndex, verifierId] of stringArray(at(task, ["dependencySmithersNodeIds"])).entries()) { - const dependency = byVerifier.get(verifierId); - const dependencyAttempt = stringField(dependency, "attemptId"); - if (dependency === undefined) { - issues.push( - issue( - `$.tasks[${taskIndex}].dependencySmithersNodeIds[${dependencyIndex}]`, - `Unknown verifier dependency ${JSON.stringify(verifierId)}` - ) - ); - } else if (dependencyAttempt !== undefined && !dependencies.includes(dependencyAttempt)) { - issues.push( - issue( - `$.tasks[${taskIndex}].dependencySmithersNodeIds[${dependencyIndex}]`, - "Verifier dependency is absent from dependency attempts" - ) - ); - } else if (dependencyAttempt !== undefined) { - joined.add(dependencyAttempt); - } - } - for (const [dependencyIndex, dependencyId] of dependencies.entries()) { - if (dependencyId === attemptId) { - issues.push( - issue(`$.tasks[${taskIndex}].dependencies[${dependencyIndex}]`, "A Smithers task cannot depend on itself") - ); - } else if (byAttempt.has(dependencyId) && !joined.has(dependencyId)) { - issues.push( - issue( - `$.tasks[${taskIndex}].dependencies[${dependencyIndex}]`, - "Task dependency is missing its verifier workflow dependency" - ) - ); - } - } - } - return issues; -} - -function smithersDependencyAcyclicityIssues(document: unknown): SemanticGateIssue[] { - const byVerifier = new Map( - smithersTasks(document).flatMap((task) => { - const id = stringField(task, "verifierSmithersNodeId"); - return id === undefined ? [] : [[id, task] as const]; - }) - ); - const visiting = new Set(); - const visited = new Set(); - let cycle: string | undefined; - const visit = (task: unknown): void => { - const id = stringField(task, "attemptId"); - if (id === undefined || cycle !== undefined || visited.has(id)) return; - if (visiting.has(id)) { - cycle = id; - return; - } - visiting.add(id); - for (const verifier of stringArray(at(task, ["dependencySmithersNodeIds"]))) { - const dependency = byVerifier.get(verifier); - if (dependency !== undefined) visit(dependency); - } - visiting.delete(id); - visited.add(id); - }; - for (const task of smithersTasks(document)) visit(task); - return cycle === undefined - ? [] - : [issue("$.tasks", `Smithers task dependencies contain a cycle at ${JSON.stringify(cycle)}`)]; -} - -function smithersPlannedAttemptIds(node: unknown): string[] { - const id = stringField(node, "id"); - if (id === undefined) return []; - const models = arrayAt(node, ["model_fanout"]); - if (models.length <= 1) return [id]; - return models.flatMap((model) => { - const modelIndex = numberField(model, "model_index"); - const attemptIndex = numberField(model, "attempt_index"); - return modelIndex === undefined || attemptIndex === undefined - ? [] - : [`${id}__model_${modelIndex}__attempt_${attemptIndex}`]; - }); -} - -function smithersGraphNodes(context: SemanticGateContext): readonly unknown[] { - return arrayAt(context.plannedGraph!.document, ["nodes"]); -} - -function smithersPlannedCoverageIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { - const tasks = smithersTasks(document); - const issues: SemanticGateIssue[] = []; - for (const [nodeIndex, node] of smithersGraphNodes(context).entries()) { - const id = stringField(node, "id"); - const matching = tasks.filter((task) => stringField(task, "concreteNodeId") === id); - if (stringField(node, "kind") === "reference") { - if (matching.length > 0) - issues.push(issue(`$.nodes[${nodeIndex}]`, "Reference planned nodes cannot have Smithers tasks")); - continue; - } - const expected = smithersPlannedAttemptIds(node); - const actual = matching.flatMap((task) => stringField(task, "attemptId") ?? "").filter((idValue) => idValue !== ""); - if (!sameStringSet(actual, expected)) { - issues.push(issue(`$.nodes[${nodeIndex}]`, `Smithers tasks do not cover planned node ${JSON.stringify(id)}`)); - } - } - return issues; -} - -function smithersPlannedIdentityIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { - const nodes = new Map( - smithersGraphNodes(context).flatMap((node) => { - const id = stringField(node, "id"); - return id === undefined ? [] : [[id, node] as const]; - }) - ); - const issues: SemanticGateIssue[] = []; - for (const [taskIndex, task] of smithersTasks(document).entries()) { - const node = nodes.get(stringField(task, "concreteNodeId") ?? ""); - if (node === undefined || stringField(node, "kind") !== "agentic") { - issues.push( - issue(`$.tasks[${taskIndex}].concreteNodeId`, "Smithers task does not join to an agentic planned node") - ); - continue; - } - const metadata = at(task, ["metadata"]); - const plannedLoop = at(node, ["loop"]); - const metadataLoop = at(metadata, ["loop"]); - if ( - stringField(task, "logicalNodeId") !== stringField(node, "logical_id") || - stringField(at(metadata, ["node"]), "logicalNodeId") !== stringField(node, "logical_id") || - stringField(at(metadata, ["node"]), "label") !== stringField(node, "display_name") || - numberField(metadataLoop, "index") !== numberField(plannedLoop, "index") || - numberField(metadataLoop, "count") !== numberField(plannedLoop, "count") || - stringField(metadataLoop, "mode") !== stringField(plannedLoop, "mode") || - numberField(metadataLoop, "attemptIndex") !== numberField(plannedLoop, "attempt_index") - ) { - issues.push(issue(`$.tasks[${taskIndex}].metadata`, "Smithers task identity does not match its planned node")); - } - const outputFields = [ - ["path", "path"], - ["contract", "contract"], - ["contractDigest", "contract_digest"], - ["schemaFile", "schema_file"], - ["schemaId", "schema_id"], - ["schemaSha256", "schema_sha256"], - ["schemaBundleSha256", "schema_bundle_sha256"], - ["validatorBuild", "validator_build"], - ["primary", "primary"] - ] as const; - const actualOutputs = arrayAt(metadata, ["artifacts", "outputs"]); - const plannedOutputs = arrayAt(node, ["outputs"]); - if ( - actualOutputs.length !== plannedOutputs.length || - plannedOutputs.some((output, outputIndex) => - outputFields.some( - ([actualField, plannedField]) => - !isRecord(actualOutputs[outputIndex]) || - !isRecord(output) || - actualOutputs[outputIndex]![actualField] !== output[plannedField] - ) - ) - ) { - issues.push( - issue( - `$.tasks[${taskIndex}].metadata.artifacts.outputs`, - "Smithers output contracts differ from the planned node" - ) - ); - } - } - return issues; -} - -function smithersPlannedDependencyJoinIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { - const nodes = new Map( - smithersGraphNodes(context).flatMap((node) => { - const id = stringField(node, "id"); - return id === undefined ? [] : [[id, node] as const]; - }) - ); - const issues: SemanticGateIssue[] = []; - for (const [taskIndex, task] of smithersTasks(document).entries()) { - const node = nodes.get(stringField(task, "concreteNodeId") ?? ""); - if (node === undefined) continue; - const dynamicDependencyIds = new Set(stringArray(at(node, ["dynamic_dependencies"]))); - const dependencyIds = stringArray(at(node, ["depends_on"])); - const pendingDynamicDependencyIds: string[] = []; - const dependencyNodes = dependencyIds.flatMap((id) => { - const dependency = nodes.get(id); - if (dependency === undefined) { - issues.push( - issue( - `$.tasks[${taskIndex}].metadata.dependencies.concreteNodeIds`, - `Smithers planned dependency node ${JSON.stringify(id)} is missing` - ) - ); - return []; - } - const dynamicStatus = at(dependency, ["dynamic", "status"]); - if (dynamicStatus === "pending" && dynamicDependencyIds.has(id)) { - pendingDynamicDependencyIds.push(id); - return []; - } - if (dynamicStatus === "expanded" && dynamicDependencyIds.has(id)) { - issues.push( - issue( - `$.tasks[${taskIndex}].metadata.dependencies.concreteNodeIds`, - `Smithers task retains expanded dynamic dependency placeholder ${JSON.stringify(id)}` - ) - ); - return []; - } - return [dependency]; - }); - const expectedAttempts = dependencyNodes.flatMap(smithersPlannedAttemptIds); - const actualAttempts = stringArray(at(task, ["dependencies"])).filter((id) => id !== "meta-start"); - if (!sameStringSet(actualAttempts, expectedAttempts)) { - issues.push( - issue(`$.tasks[${taskIndex}].dependencies`, "Smithers dependency attempts do not match planned dependencies") - ); - } - const expectedNodes = dependencyNodes.flatMap((dependency) => { - const id = stringField(dependency, "id"); - return id === undefined ? [] : [id]; - }); - const actualNodes = stringArray(at(task, ["metadata", "dependencies", "concreteNodeIds"])).filter( - (id) => id !== "__start__" - ); - const compiledExpectedNodes = [...expectedNodes, ...pendingDynamicDependencyIds]; - if (!sameStringSet(actualNodes, expectedNodes) && !sameStringSet(actualNodes, compiledExpectedNodes)) { - issues.push( - issue( - `$.tasks[${taskIndex}].metadata.dependencies.concreteNodeIds`, - "Smithers concrete dependencies do not match the plan" - ) - ); - } - const expectedVerifiers = dependencyNodes - .filter((dependency) => stringField(dependency, "kind") === "agentic") - .flatMap(smithersPlannedAttemptIds) - .map((id) => `verify:${id}`); - if (!sameStringSet(stringArray(at(task, ["dependencySmithersNodeIds"])), expectedVerifiers)) { - issues.push( - issue(`$.tasks[${taskIndex}].dependencySmithersNodeIds`, "Smithers workflow dependencies do not match the plan") - ); - } - } - return issues; -} - function workspacePatchPathIssues(document: unknown): SemanticGateIssue[] { const included = arrayAt(document, ["files"]); const excluded = arrayAt(document, ["excluded_files"]); @@ -5797,73 +5087,6 @@ function resolveArtifactFile(rootDirectory: string, relativePath: string): strin return candidate !== root && candidate.startsWith(`${root}${path.sep}`) ? candidate : undefined; } -function sha256File(filePath: string): string { - return crypto.createHash("sha256").update(fs.readFileSync(filePath)).digest("hex"); -} - -function filesystemManifestIssues( - document: unknown, - context: SemanticGateContext, - rowsPath: readonly string[] -): SemanticGateIssue[] { - const root = context.filesystem!.rootDirectory; - const issues: SemanticGateIssue[] = []; - for (const [index, row] of arrayAt(document, rowsPath).entries()) { - const relativePath = stringField(row, "path"); - const expectedDigest = stringField(row, "sha256"); - if (relativePath === undefined) continue; - const filePath = resolveArtifactFile(root, relativePath); - const snapshot = context.filesystem!.files?.get(relativePath); - if (context.filesystem!.files !== undefined) { - const rowPath = `${displayPath(rowsPath)}[${index}]`; - if (filePath === undefined || snapshot === undefined) { - issues.push( - issue(`${rowPath}.path`, `Referenced file is missing or nonregular: ${JSON.stringify(relativePath)}`) - ); - continue; - } - if ( - expectedDigest !== undefined && - crypto.createHash("sha256").update(snapshot).digest("hex") !== expectedDigest - ) { - issues.push( - issue(`${rowPath}.sha256`, `Referenced file digest does not match ${JSON.stringify(relativePath)}`) - ); - } - const expectedSize = numberField(row, "size_bytes"); - if (expectedSize !== undefined && expectedSize !== snapshot.byteLength) { - issues.push( - issue(`${rowPath}.size_bytes`, `Referenced file size does not match ${JSON.stringify(relativePath)}`) - ); - } - continue; - } - let stats: fs.Stats | undefined; - try { - if (filePath !== undefined) stats = fs.lstatSync(filePath); - } catch { - // Reported below as a missing/nonregular file. - } - const rowPath = `${displayPath(rowsPath)}[${index}]`; - if (filePath === undefined || stats === undefined || !stats.isFile() || stats.isSymbolicLink()) { - issues.push( - issue(`${rowPath}.path`, `Referenced file is missing or nonregular: ${JSON.stringify(relativePath)}`) - ); - continue; - } - if (expectedDigest !== undefined && sha256File(filePath) !== expectedDigest) { - issues.push(issue(`${rowPath}.sha256`, `Referenced file digest does not match ${JSON.stringify(relativePath)}`)); - } - const expectedSize = numberField(row, "size_bytes"); - if (expectedSize !== undefined && expectedSize !== stats.size) { - issues.push( - issue(`${rowPath}.size_bytes`, `Referenced file size does not match ${JSON.stringify(relativePath)}`) - ); - } - } - return issues; -} - function generatedTestFileIntegrityIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { const root = context.filesystem!.rootDirectory; const resourceIssues = generatedTestBundleResourceBoundsIssues(document); @@ -6090,94 +5313,6 @@ function generatedTestIdentityIssues(document: unknown, context: SemanticGateCon return issues; } -function agentSourceProofGitIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { - const git = context.git!; - const issues: SemanticGateIssue[] = []; - if (stringField(document, "commit") !== git.commit) - issues.push(issue("$.commit", "Source proof commit does not match Git")); - if (stringField(document, "tree") !== git.tree) issues.push(issue("$.tree", "Source proof tree does not match Git")); - for (const [index, ref] of arrayAt(document, ["refs"]).entries()) { - const name = stringField(ref, "name"); - const object = stringField(ref, "object"); - if (name !== undefined && git.refs?.[name] !== object) { - issues.push(issue(`$.refs[${index}].object`, `Source proof ref ${JSON.stringify(name)} does not match Git`)); - } - } - return issues; -} - -function agentSourceProofDependencyIssues(document: unknown): SemanticGateIssue[] { - const dependencies = at(document, ["dependencies"]); - if (dependencies === null || !isRecord(dependencies)) return []; - const issues: SemanticGateIssue[] = []; - issues.push(...pinnedSubmodulePortablePathIssues(dependencies, "$.dependencies")); - if ( - stringField(dependencies, "source_commit") !== stringField(document, "commit") || - stringField(dependencies, "source_tree") !== stringField(document, "tree") - ) { - issues.push(issue("$.dependencies", "Pinned dependency source identity does not match the source proof")); - } - const roots = stringArray(at(dependencies, ["top_level_roots"])); - const gitlinks = arrayAt(dependencies, ["recursive_gitlinks"]); - const gitlinkPaths = gitlinks.flatMap((entry) => { - const entryPath = stringField(entry, "path"); - return entryPath === undefined ? [] : [entryPath]; - }); - const canonical = (values: readonly string[]): boolean => - new Set(values).size === values.length && values.every((value, index) => index === 0 || values[index - 1]! < value); - if (!canonical(roots)) { - issues.push(issue("$.dependencies.top_level_roots", "Pinned dependency roots must be unique and ordered")); - } - if (!canonical(gitlinkPaths)) { - issues.push(issue("$.dependencies.recursive_gitlinks", "Pinned dependency gitlinks must be unique and ordered")); - } - for (const [index, root] of roots.entries()) { - if (!gitlinkPaths.includes(root)) { - issues.push(issue(`$.dependencies.top_level_roots[${index}]`, "Pinned dependency root is not a gitlink")); - } - if (roots.some((candidate, candidateIndex) => candidateIndex !== index && root.startsWith(`${candidate}/`))) { - issues.push(issue(`$.dependencies.top_level_roots[${index}]`, "Pinned dependency roots overlap")); - } - } - const entryCount = numberField(dependencies, "entry_count"); - const fileCount = numberField(dependencies, "file_count"); - if (entryCount !== undefined && fileCount !== undefined && fileCount > entryCount) { - issues.push(issue("$.dependencies.file_count", "Pinned dependency file count exceeds entry count")); - } - return issues; -} - -function pinnedSubmodulePortablePathIssues(expectation: unknown, basePath: string): SemanticGateIssue[] { - const candidates = [ - ...stringArray(at(expectation, ["top_level_roots"])).map((value, index) => ({ - value, - path: `${basePath}.top_level_roots[${index}]` - })), - ...arrayAt(expectation, ["recursive_gitlinks"]).flatMap((entry, index) => { - const value = stringField(entry, "path"); - return value === undefined ? [] : [{ value, path: `${basePath}.recursive_gitlinks[${index}].path` }]; - }) - ]; - return candidates.flatMap(({ value, path: issuePath }) => { - const segments = value.split("/"); - return value.includes("\\") || - path.posix.isAbsolute(value) || - path.posix.normalize(value) !== value || - Buffer.byteLength(value, "utf8") > 4_096 || - segments.length > 128 || - segments.some( - (segment) => - segment.length === 0 || - segment === "." || - segment === ".." || - segment === ".git" || - /^[A-Za-z]:/u.test(segment) - ) - ? [issue(issuePath, "Pinned submodule path is not portable and bounded")] - : []; - }); -} - function invariantSourceProofGitIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { const git = context.git!; const issues: SemanticGateIssue[] = []; @@ -6208,38 +5343,6 @@ function workspacePatchGitIssues(document: unknown, context: SemanticGateContext ); } -function artifactVerificationPlanIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { - const planned = context.plannedGraph!.node; - const issues: SemanticGateIssue[] = []; - if (stringField(document, "node_id") !== stringField(planned, "id")) { - issues.push(issue("$.node_id", "Verification marker node_id does not match the planned node")); - } - const actual = arrayAt(document, ["artifacts"]); - const expected = arrayAt(planned, ["outputs"]); - if (actual.length !== expected.length) { - issues.push(issue("$.artifacts", "Verification marker artifact count does not match planned outputs")); - return issues; - } - for (const [index, output] of expected.entries()) { - const artifact = actual[index]; - for (const field of [ - "path", - "contract", - "contract_digest", - "schema_file", - "schema_id", - "schema_sha256", - "schema_bundle_sha256", - "validator_build", - "primary" - ] as const) { - if (isRecord(artifact) && isRecord(output) && artifact[field] === output[field]) continue; - issues.push(issue(`$.artifacts[${index}].${field}`, `Verification marker ${field} does not match the plan`)); - } - } - return issues; -} - function campaignSummaryCountIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { const campaigns = context.artifactSet!.campaigns!; const findings = context.artifactSet!.findings!; @@ -6549,10 +5652,11 @@ function propertyCampaignDocumentIssues(document: unknown): SemanticGateIssue[] } const RECON_MAX_TEST_LIMIT = "18446744073709551615"; +const RECON_STATEFUL_SEQUENCE_LENGTH = 100; const CAMPAIGN_HOST_FORCE_KILL_GRACE_SECONDS = 300; const CAMPAIGN_DURATION_TOLERANCE_MS = 5_000; -function campaignTimeoutFlagValues(command: string, flag: "--timeout" | "--test-limit"): string[] { +function campaignTimeoutFlagValues(command: string, flag: "--timeout" | "--test-limit" | "--seq-len"): string[] { const escapedFlag = flag.replace(/[.*+?^${}()|[\]\\]/gu, "\\$&"); const pattern = new RegExp(`(?:^|\\s)${escapedFlag}(?:(?:=|\\s+)(\\S+))?`, "gu"); return [...command.matchAll(pattern)].map((match) => match[1] ?? ""); @@ -6685,6 +5789,27 @@ function propertyCampaignTimeoutEvidenceIssues(document: unknown, context: Seman RECON_MAX_TEST_LIMIT, "recon_test_limit must use the nonbinding maximum" ); + compare( + `${planPath}#recon_sequence_length`, + numberField(plan, "recon_sequence_length"), + RECON_STATEFUL_SEQUENCE_LENGTH, + "recon_sequence_length must be the stateful campaign sequence length" + ); + compare( + `${summaryPath}#sequence_length`, + numberField(summary, "sequence_length"), + RECON_STATEFUL_SEQUENCE_LENGTH, + "Campaign summary sequence_length must be the stateful campaign sequence length" + ); + // Optional in the result schema, so it is checked only when recorded. + if (isRecord(document) && document.sequence_length !== undefined) { + compare( + "$.sequence_length", + numberField(document, "sequence_length"), + RECON_STATEFUL_SEQUENCE_LENGTH, + "sequence_length must be the stateful campaign sequence length" + ); + } const backend = at(plan, ["backend"]); const campaignCommands = arrayAt(plan, ["command_plan"]).filter((row) => stringField(row, "phase") === "campaign"); @@ -6715,6 +5840,7 @@ function propertyCampaignTimeoutEvidenceIssues(document: unknown, context: Seman if (resultCommand !== undefined) { const timeoutValues = campaignTimeoutFlagValues(resultCommand, "--timeout"); const testLimitValues = campaignTimeoutFlagValues(resultCommand, "--test-limit"); + const sequenceLengthValues = campaignTimeoutFlagValues(resultCommand, "--seq-len"); if (timeoutValues.length !== 1 || timeoutValues[0] !== String(configuredTimeoutSeconds)) { issues.push( issue("$.exact_command", `Recon command must contain exactly one --timeout ${configuredTimeoutSeconds} flag`) @@ -6725,6 +5851,10 @@ function propertyCampaignTimeoutEvidenceIssues(document: unknown, context: Seman issue("$.exact_command", `Recon command must contain exactly one --test-limit ${RECON_MAX_TEST_LIMIT} flag`) ); } + const sequenceLength = String(RECON_STATEFUL_SEQUENCE_LENGTH); + if (sequenceLengthValues.length !== 1 || sequenceLengthValues[0] !== sequenceLength) { + issues.push(issue("$.exact_command", `Recon command must contain exactly one --seq-len ${sequenceLength} flag`)); + } if (!hasExactCampaignHostTimeoutWrapper(resultCommand, configuredTimeoutSeconds)) { issues.push( issue( @@ -7418,18 +6548,6 @@ function attemptSourceEventJoinIssues(document: unknown, context: SemanticGateCo return issues; } -function runStateFingerprintIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { - const expected = context.runtimeState!; - const issues: SemanticGateIssue[] = []; - if (stringField(document, "graph_fingerprint") !== expected.graphFingerprint) { - issues.push(issue("$.graph_fingerprint", "Run-state graph fingerprint does not match runtime state")); - } - if (stringField(document, "config_fingerprint") !== expected.configFingerprint) { - issues.push(issue("$.config_fingerprint", "Run-state config fingerprint does not match runtime state")); - } - return issues; -} - function usageEventOrderIssues(document: unknown, context: SemanticGateContext): SemanticGateIssue[] { const entries = [...context.usageLedger!.entries]; if (!entries.includes(document)) entries.push(document); @@ -7512,13 +6630,6 @@ const gateSpecifications = { "selected-strategies-metadata-completeness": documentGate(artifactMetadataCompletenessIssues), "admin-config-surface-id-uniqueness": documentGate(uniqueFieldGate([["surfaces"]], "surface_id", "admin surface ID")), "admin-config-surface-joins": documentGate(adminConfigJoinIssues), - "agent-source-proof-commit-binding": contextualGate( - "git", - ["git.commit", "git.tree", "git.refs"], - agentSourceProofGitIssues - ), - "agent-source-proof-dependency-lineage": documentGate(agentSourceProofDependencyIssues), - "agent-source-proof-ref-uniqueness": documentGate(uniqueFieldGate([["refs"]], "name", "source proof ref name")), "aggregation-count-coupling": documentGate(aggregationCountIssues), "aggregation-source-bundle-reconciliation": documentGate(aggregationBundleIssues), "aggregation-resource-bounds": documentGate(aggregationResourceIssues), @@ -7529,9 +6640,6 @@ const gateSpecifications = { ), "aggregation-destination-path-uniqueness": documentGate(aggregationDestinationIssues), "aggregation-source-entry-uniqueness": documentGate(aggregationSourceEntryIssues), - "analysis-bundle-file-digest": contextualGate("filesystem", ["filesystem.rootDirectory"], (document, context) => - filesystemManifestIssues(document, context, ["files"]) - ), "analysis-bundle-accounting-reconciliation": documentGate(analysisBundleAccountingIssues), "analysis-bundle-attempt-order": documentGate(analysisBundleAttemptOrderIssues), "analysis-bundle-evaluation-count-reconciliation": documentGate(analysisBundleEvaluationCountIssues), @@ -7544,31 +6652,6 @@ const gateSpecifications = { "analysis-bundle-path-order": documentGate(analysisBundlePathIssues), "analysis-bundle-recovery-reconciliation": documentGate(analysisBundleRecoveryIssues), "analysis-bundle-terminal-status-reconciliation": documentGate(analysisBundleTerminalStatusIssues), - "artifact-manifest-file-digest": contextualGate("filesystem", ["filesystem.rootDirectory"], (document, context) => - filesystemManifestIssues(document, context, ["files"]) - ), - "artifact-manifest-file-path-uniqueness": documentGate(uniqueFieldGate([["files"]], "path", "artifact file path")), - "artifact-manifest-output-path-uniqueness": documentGate( - uniqueFieldGate([["output_contracts"]], "path", "artifact output path") - ), - "artifact-manifest-prerequisite-node-uniqueness": documentGate( - uniqueFieldGate([["prerequisite_manifests"]], "node_id", "prerequisite node ID") - ), - "artifact-verification-artifact-path-uniqueness": documentGate( - uniqueFieldGate([["artifacts"]], "path", "verified artifact path") - ), - "artifact-verification-exactly-one-primary": documentGate((document) => - exactlyOnePrimaryIssues(document, "artifacts") - ), - "artifact-verification-plan-contract-identity": contextualGate( - "cross-artifact", - ["plannedGraph.node"], - artifactVerificationPlanIssues - ), - "artifact-verification-publication-digest-correspondence": documentGate(artifactVerificationDigestIssues), - "artifact-verification-publication-path-uniqueness": documentGate( - uniqueFieldGate([["publications"]], "path", "publication path") - ), "attempt-order": documentGate(attemptOrderIssues), "attempt-failure-message-byte-length": documentGate(attemptFailureMessageByteLengthIssues), "attempt-outcome-digest-coupling": documentGate(attemptOutcomeDigestIssues), @@ -7597,27 +6680,6 @@ const gateSpecifications = { ["artifactSet.campaigns", "artifactSet.findings"], campaignSummaryCountIssues ), - "config-redactions-path-key-equality": documentGate((document) => - arrayAt(document, ["entries"]).flatMap((entry, index) => { - const pathValue = at(entry, ["path"]); - const projected = Array.isArray(pathValue) ? pathValue.join(".") : undefined; - const key = stringField(entry, "key"); - return projected !== undefined && key !== projected - ? [issue(`$.entries[${index}].key`, "Configuration redaction key does not equal its projected path")] - : []; - }) - ), - "config-redactions-path-uniqueness": documentGate((document) => { - const seen = new Set(); - return arrayAt(document, ["entries"]).flatMap((entry, index) => { - const pathValue = at(entry, ["path"]); - if (!Array.isArray(pathValue)) return []; - const projected = JSON.stringify(pathValue); - const duplicate = seen.has(projected); - seen.add(projected); - return duplicate ? [issue(`$.entries[${index}].path`, "Duplicate configuration redaction path")] : []; - }); - }), "dependency-id-uniqueness": documentGate(uniqueFieldGate([["dependencies"]], "dependency_id", "dependency ID")), "dependency-row-joins": documentGate(dependencyJoinIssues), "differential-gap-review-lane-reconciliation": contextualGate( @@ -7737,9 +6799,6 @@ const gateSpecifications = { ["artifactSet.reviewStage"], lifecycleReviewStageIssues ), - "finding-evidence-span-consistency": documentGate((document) => findingEvidenceSpanIssues(document)), - "finding-campaign-provenance-coherence": documentGate((document) => findingCampaignProvenanceIssues(document)), - "finding-projected-reference-uniqueness": documentGate(findingProjectedReferenceIssues), "findings-campaign-provenance-coherence": documentGate(findingArrayCampaignProvenanceIssues), "findings-evidence-span-consistency": documentGate(findingArrayEvidenceSpanIssues), "findings-id-uniqueness": documentGate(uniqueFieldGate([[]], "id", "finding ID")), @@ -7775,56 +6834,16 @@ const gateSpecifications = { "invariant-source-proof-path-uniqueness": documentGate( uniqueFieldGate([["files"]], "path", "invariant source proof path") ), - "invariant-suite-file-path-uniqueness": documentGate( - uniqueFieldGate([["files"]], "path", "invariant suite file path") - ), - "invariant-suite-file-tombstone-disjointness": documentGate((document) => { - const files = new Set(arrayAt(document, ["files"]).flatMap((row) => stringField(row, "path") ?? "")); - return stringArray(at(document, ["tombstones"])).flatMap((tombstone, index) => - files.has(tombstone) - ? [ - issue( - `$.tombstones[${index}]`, - `Invariant suite path is both present and tombstoned ${JSON.stringify(tombstone)}` - ) - ] - : [] - ); - }), - "invariant-suite-tombstone-uniqueness": documentGate((document) => { - const tombstones = stringArray(at(document, ["tombstones"])); - const seen = new Set(); - return tombstones.flatMap((tombstone, index) => { - const duplicate = seen.has(tombstone); - seen.add(tombstone); - return duplicate - ? [issue(`$.tombstones[${index}]`, `Duplicate invariant suite tombstone ${JSON.stringify(tombstone)}`)] - : []; - }); - }), "json-validator-preflight-current-identity": contextualGate( "runtime-state", [ "validatorPreflight.schemaId", "validatorPreflight.schemaSha256", "validatorPreflight.schemaBundleSha256", - "validatorPreflight.validatorBuild", "validatorPreflight.artifactSha256" ], jsonValidatorPreflightIdentityIssues ), - "planned-graph-acyclicity": documentGate(plannedAcyclicityIssues), - "planned-graph-artifact-dir-identity": documentGate(plannedArtifactDirIssues), - "planned-graph-contract-identity": documentGate(plannedContractIdentityIssues), - "planned-graph-dependency-join": documentGate(plannedDependencyJoinIssues), - "planned-graph-exactly-one-primary": documentGate(plannedPrimaryIssues), - "planned-graph-loop-coupling": documentGate(plannedLoopIssues), - "planned-graph-model-fanout-uniqueness": documentGate(plannedModelFanoutIssues), - "planned-graph-model-loop-coupling": documentGate(plannedModelLoopIssues), - "planned-graph-node-id-uniqueness": documentGate(plannedNodeIdIssues), - "planned-graph-output-path-uniqueness": documentGate(plannedOutputPathIssues), - "planned-graph-workflow-node-join": documentGate(plannedWorkflowJoinIssues), - "planned-graph-workflow-task-uniqueness": documentGate(plannedWorkflowTaskIssues), "property-campaign-coverage-metric-uniqueness": documentGate( uniqueFieldGate([["coverage", "metrics"]], "name", "property campaign coverage metric") ), @@ -7920,43 +6939,6 @@ const gateSpecifications = { ["artifactSet.propertyCatalog", "artifactSet.implementedProperties"], reportPropertyJoinIssues ), - "run-metadata-accounting-workflow-identity": documentGate((document) => { - const workflowRunId = stringField(at(document, ["workflow"]), "run_id"); - const accounting = at(document, ["accounting"]); - if (accounting === undefined) return []; - const accountingRunId = stringField(accounting, "workflow_run_id"); - const currentRunId = stringField(at(accounting, ["current"]), "workflow_run_id"); - return workflowRunId !== accountingRunId || workflowRunId !== currentRunId - ? [issue("$.accounting.workflow_run_id", "Accounting identity does not equal the active workflow run")] - : []; - }), - "run-metadata-current-segment-equality": documentGate((document) => { - const accounting = at(document, ["accounting"]); - if (accounting === undefined) return []; - const segments = arrayAt(accounting, ["segments"]); - return isDeepStrictEqual(at(accounting, ["current"]), segments.at(-1)) - ? [] - : [issue("$.accounting.current", "Current accounting does not equal the final segment")]; - }), - "run-metadata-workflow-id-equality": documentGate((document) => { - const workflow = at(document, ["workflow"]); - const ids = stringArray(at(document, ["workflow_ids"])); - if (workflow === undefined) { - return ids.length === 0 ? [] : [issue("$.workflow_ids", "Unlinked metadata carries workflow IDs")]; - } - const runId = stringField(workflow, "run_id"); - return ids.length === 1 && ids[0] === runId - ? [] - : [issue("$.workflow_ids", "Workflow IDs do not equal the active workflow run")]; - }), - "run-plan-attempt-id-uniqueness": documentGate( - uniqueFieldGate([["rendered_prompts"]], "attempt_id", "rendered prompt attempt ID") - ), - "run-state-fingerprint": contextualGate( - "runtime-state", - ["runtimeState.graphFingerprint", "runtimeState.configFingerprint"], - runStateFingerprintIssues - ), "run-state-node-key-equality": documentGate(runStateNodeKeyIssues), "selected-strategy-id-uniqueness": documentGate( uniqueFieldGate([["strategies"]], "strategy_id", "selected strategy ID") @@ -7978,32 +6960,6 @@ const gateSpecifications = { ["artifactSet.triagedFindings"], severityClassificationPreservationIssues ), - "smithers-task-attempt-id-uniqueness": documentGate(uniqueFieldGate([["tasks"]], "attemptId", "Smithers attempt ID")), - "smithers-task-workflow-id-uniqueness": documentGate(smithersWorkflowIdentityIssues), - "smithers-task-document-identity": documentGate(smithersDocumentIdentityIssues), - "smithers-task-pinned-submodule-expectation": documentGate(smithersPinnedSubmoduleIssues), - "smithers-task-dependency-join": documentGate(smithersDependencyJoinIssues), - "smithers-task-dependency-acyclicity": documentGate(smithersDependencyAcyclicityIssues), - "smithers-task-planned-graph-coverage": contextualGate( - "cross-artifact", - ["plannedGraph.document"], - smithersPlannedCoverageIssues - ), - "smithers-task-planned-graph-identity": contextualGate( - "cross-artifact", - ["plannedGraph.document"], - smithersPlannedIdentityIssues - ), - "smithers-task-planned-graph-dependency-join": contextualGate( - "cross-artifact", - ["plannedGraph.document"], - smithersPlannedDependencyJoinIssues - ), - "source-run-not-self": documentGate((document) => - stringField(document, "run_id") === stringField(document, "source_run_id") - ? [issue("$.source_run_id", "Source run must differ from the destination run")] - : [] - ), "semantic-red-registry-lane-reconciliation": contextualGate( "cross-artifact", ["artifactSet.differentialArtifacts.laneResults"], diff --git a/packages/artifacts/src/smithers-task-manifest.ts b/packages/artifacts/src/smithers-task-manifest.ts index 8dccba01c..3744f845b 100644 --- a/packages/artifacts/src/smithers-task-manifest.ts +++ b/packages/artifacts/src/smithers-task-manifest.ts @@ -1410,7 +1410,11 @@ export function assertSmithersTaskManifestMatchesPlannedGraph( } } - const optionalAttemptIds = new Set( + // Safety only: a producer that halts on failure can never have its output treated as optional. + // Which consumers opt in to a continuing producer's output is compiler policy and is not + // re-derived here: once a second copy of that policy drifted from the compiler, it rejected the + // default, low-cost, exhaustive and invariant-only launches (#1140). + const continuingAttemptIds = new Set( manifest.tasks .filter((task) => { const node = graphNodes.get(task.concreteNodeId)!; @@ -1419,21 +1423,14 @@ export function assertSmithersTaskManifestMatchesPlannedGraph( .map((task) => task.attemptId) ); for (const task of manifest.tasks) { - // Mirrors the runtime compiler: continuation lets independent tasks settle, - // but only the review group reconciles partial results, so only review tasks - // treat inputs from continuing groups as optional. - const reconcilesPartialResults = graphNodes.get(task.concreteNodeId)?.group === "review"; - const expectedOptionalDirectories = reconcilesPartialResults - ? task.dependencyArtifactDirs.filter((directory) => { - const attemptId = directory.split(/[\\/]/u).at(-1); - return attemptId !== undefined && optionalAttemptIds.has(attemptId); - }) - : []; - assertSameStringSet( - task.optionalDependencyArtifactDirs ?? [], - expectedOptionalDirectories, - `Smithers task ${JSON.stringify(task.attemptId)} optional dependency artifact directories` - ); + for (const directory of task.optionalDependencyArtifactDirs ?? []) { + const producer = portablePathBasename(directory) ?? ""; + if (!continuingAttemptIds.has(producer)) { + throw new Error( + `Smithers task ${JSON.stringify(task.attemptId)} marks dependency ${JSON.stringify(producer)} optional, but that producer does not continue on failure` + ); + } + } } } diff --git a/packages/artifacts/src/state-schema.ts b/packages/artifacts/src/state-schema.ts index bf1a9aaf4..61714ae13 100644 --- a/packages/artifacts/src/state-schema.ts +++ b/packages/artifacts/src/state-schema.ts @@ -23,8 +23,7 @@ import { TERMINAL_NODE_STATE_STATUSES, isTerminalNodeStatus, type NodeState, - type RunState, - type TerminalDispositionDocument + type RunState } from "./state.js"; import { schemaErrorMessage, validateWithZod, type SchemaValidationResult } from "./schema-validation.js"; @@ -825,16 +824,6 @@ export function validateRunStateSchema(value: unknown, path = "$"): SchemaValida }); } -export function validateTerminalDispositionSchema( - value: unknown, - path = "$" -): SchemaValidationResult { - return validateWithZod(terminalDispositionSchema as z.ZodType, value, { - path, - code: "TERMINAL_DISPOSITION_SCHEMA_INVALID" - }); -} - export function validateNodeStateSchema(value: unknown, path = "$"): SchemaValidationResult { return validateWithZod(nodeStateSchema as z.ZodType, value, { path, @@ -849,11 +838,3 @@ export function assertRunStateSchema(value: unknown): RunState { } return result.value; } - -export function assertNodeStateSchema(value: unknown): NodeState { - const result = validateNodeStateSchema(value); - if (!result.ok || !result.value) { - throw new Error(schemaErrorMessage("node state", result.issues)); - } - return result.value; -} diff --git a/packages/artifacts/src/state.ts b/packages/artifacts/src/state.ts index f79aa0e2d..b370d88b5 100644 --- a/packages/artifacts/src/state.ts +++ b/packages/artifacts/src/state.ts @@ -1,5 +1,3 @@ -import fs from "node:fs"; - import { redactSecretsInText, SENSITIVE_REDACTION_PLACEHOLDER } from "@ultrafuzz/security"; import type { ArtifactContractId } from "./artifact-contract-ids.js"; @@ -557,19 +555,6 @@ function assertCurrentRunState(value: unknown): asserts value is RunState { } } -export function loadOrCreateRunState( - target: RunLayoutStateLike | string, - state: RunState, - options: RunStateWriteOptions = {} -): RunState { - const statePath = resolveStatePath(target); - if (fs.existsSync(statePath)) { - return readRunState(statePath); - } - writeRunState(statePath, state, options); - return state; -} - export function updateRunStatus( target: RunLayoutStateLike | string, status: RunStatus, diff --git a/packages/artifacts/src/strict-jsonl.ts b/packages/artifacts/src/strict-jsonl.ts index 3afea1ce4..255df935f 100644 --- a/packages/artifacts/src/strict-jsonl.ts +++ b/packages/artifacts/src/strict-jsonl.ts @@ -33,15 +33,18 @@ export function readStrictJsonlSnapshot( filePath: string, codec: StrictJsonlCodec ): StrictJsonlSnapshot { + const bytes = readStrictJsonlBytes(filePath, codec); + return bytes === undefined ? { records: [], byteLength: 0, exists: false } : parseStrictJsonlBytes(bytes, codec); +} + +function readStrictJsonlBytes(filePath: string, codec: StrictJsonlCodec): Buffer | undefined { try { fs.lstatSync(filePath); } catch (error) { - if (isErrnoException(error, "ENOENT")) return { records: [], byteLength: 0, exists: false }; + if (isErrnoException(error, "ENOENT")) return undefined; throw new Error(`failed to inspect ${codec.label} ${filePath}`, { cause: error }); } - const maxBytes = codec.maxBytes ?? DEFAULT_STRICT_JSONL_MAX_BYTES; - const bytes = readRegularFileSnapshot(filePath, maxBytes); - return parseStrictJsonlBytes(bytes, codec); + return readRegularFileSnapshot(filePath, codec.maxBytes ?? DEFAULT_STRICT_JSONL_MAX_BYTES); } /** Parse one already-captured immutable JSONL byte snapshot. */ @@ -72,35 +75,42 @@ export function parseStrictJsonlBytes( throw new Error(`${codec.label} exceeds the ${maxRecords}-record limit`); } - const maxRecordBytes = codec.maxRecordBytes ?? DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES; - const records: RecordType[] = []; - for (const [index, line] of lines.entries()) { - const lineNumber = index + 1; - if (line.trim().length === 0) throw new Error(`${codec.label} contains a blank record at line ${lineNumber}`); - const lineBytes = Buffer.from(line, "utf8"); - if (lineBytes.byteLength > maxRecordBytes) { - throw new Error(`${codec.label} record ${lineNumber} exceeds the ${maxRecordBytes}-byte limit`); - } - let parsed: unknown; - try { - parsed = parseStrictJsonBytes(lineBytes, { - maxBytes: maxRecordBytes, - maxDepth: 128, - maxItems: 100_000, - maxProperties: 100_000 - }); - } catch (error) { - throw new Error( - `${codec.label} record ${lineNumber} is invalid strict JSON: ${error instanceof Error ? error.message : String(error)}`, - { cause: error } - ); - } - records.push(codec.parseRecord(parsed, `$[${index}]`)); - } + const records = lines.map((line, index) => + parseStrictJsonlLine(line, Buffer.from(line, "utf8"), String(index + 1), `$[${String(index)}]`, codec) + ); validateStrictJsonlHistory(records, codec); return { records, byteLength: bytes.byteLength, exists: true }; } +function parseStrictJsonlLine( + line: string, + lineBytes: Buffer, + lineLabel: string, + recordPath: string, + codec: StrictJsonlCodec +): RecordType { + if (line.trim().length === 0) throw new Error(`${codec.label} contains a blank record at line ${lineLabel}`); + const maxRecordBytes = codec.maxRecordBytes ?? DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES; + if (lineBytes.byteLength > maxRecordBytes) { + throw new Error(`${codec.label} record ${lineLabel} exceeds the ${String(maxRecordBytes)}-byte limit`); + } + let parsed: unknown; + try { + parsed = parseStrictJsonBytes(lineBytes, { + maxBytes: maxRecordBytes, + maxDepth: 128, + maxItems: 100_000, + maxProperties: 100_000 + }); + } catch (error) { + throw new Error( + `${codec.label} record ${lineLabel} is invalid strict JSON: ${error instanceof Error ? error.message : String(error)}`, + { cause: error } + ); + } + return codec.parseRecord(parsed, recordPath); +} + function isErrnoException(error: unknown, code: string): error is NodeJS.ErrnoException { return error instanceof Error && "code" in error && (error as NodeJS.ErrnoException).code === code; } @@ -138,7 +148,84 @@ export function appendStrictJsonlRecords( const existing = readStrictJsonlSnapshot(filePath, codec); if (records.length === 0) return existing; - const canonicalRecords = records.map((record, index) => { + const canonicalRecords = canonicalStrictJsonlRecords( + records, + codec, + (index) => `$[${String(existing.records.length + index)}]` + ); + const combined = [...existing.records, ...canonicalRecords]; + if (combined.length > (codec.maxRecords ?? DEFAULT_STRICT_JSONL_MAX_RECORDS)) { + throw new Error(`${codec.label} exceeds the record limit`); + } + validateStrictJsonlHistory(combined, codec); + const byteLength = writeStrictJsonlRecords(filePath, existing, canonicalRecords, codec, trustedRoot); + // The write is fenced at the snapshot length, so the file now holds exactly these records. + return { records: combined, byteLength, exists: true }; +} + +/** + * Append records after validating them against only the journal's trailing + * records: the final record, then earlier ones while `inWindow` holds for + * them. This suits journals whose history rules relate a record only to the + * records inside that window. It does not count records, so it is for + * journals bounded by bytes alone. Readers still validate the whole journal, + * and the byte length read here still fences the write. + */ +export function appendStrictJsonlRecordsAfterTail( + filePath: string, + records: readonly RecordType[], + codec: StrictJsonlCodec, + inWindow: (existing: RecordType) => boolean, + trustedRoot?: string +): void { + if (records.length === 0) return; + const bytes = readStrictJsonlBytes(filePath, codec); + const tail = bytes === undefined ? [] : parseStrictJsonlTail(bytes, codec, inWindow); + const canonicalRecords = canonicalStrictJsonlRecords(records, codec, (index) => `$[new ${String(index)}]`); + validateStrictJsonlHistory([...tail, ...canonicalRecords], codec); + writeStrictJsonlRecords( + filePath, + { exists: bytes !== undefined, byteLength: bytes?.byteLength ?? 0 }, + canonicalRecords, + codec, + trustedRoot + ); +} + +function parseStrictJsonlTail( + bytes: Buffer, + codec: StrictJsonlCodec, + inWindow: (existing: RecordType) => boolean +): RecordType[] { + if (bytes.byteLength === 0) return []; + if (bytes[bytes.byteLength - 1] !== 0x0a) { + throw new Error(`${codec.label} has a torn or unterminated final record`); + } + const tail: RecordType[] = []; + for (let end = bytes.byteLength - 1; end >= 0;) { + const start = end === 0 ? 0 : bytes.lastIndexOf(0x0a, end - 1) + 1; + const lineBytes = bytes.subarray(start, end); + const fromEnd = tail.length + 1; + const record = parseStrictJsonlLine( + lineBytes.toString("utf8"), + lineBytes, + `${String(fromEnd)} from the end`, + `$[-${String(fromEnd)}]`, + codec + ); + tail.unshift(record); + if (!inWindow(record)) break; + end = start - 1; + } + return tail; +} + +function canonicalStrictJsonlRecords( + records: readonly RecordType[], + codec: StrictJsonlCodec, + recordPath: (index: number) => string +): RecordType[] { + return records.map((record, index) => { const serialized = JSON.stringify(record); const parsed = parseStrictJsonBytes(Buffer.from(serialized, "utf8"), { maxBytes: codec.maxRecordBytes ?? DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES, @@ -146,14 +233,17 @@ export function appendStrictJsonlRecords( maxItems: 100_000, maxProperties: 100_000 }); - return codec.parseRecord(parsed, `$[${existing.records.length + index}]`); + return codec.parseRecord(parsed, recordPath(index)); }); - const combined = [...existing.records, ...canonicalRecords]; - if (combined.length > (codec.maxRecords ?? DEFAULT_STRICT_JSONL_MAX_RECORDS)) { - throw new Error(`${codec.label} exceeds the record limit`); - } - validateStrictJsonlHistory(combined, codec); +} +function writeStrictJsonlRecords( + filePath: string, + existing: { exists: boolean; byteLength: number }, + canonicalRecords: readonly RecordType[], + codec: StrictJsonlCodec, + trustedRoot: string | undefined +): number { const payload = Buffer.from(`${canonicalRecords.map((record) => JSON.stringify(record)).join("\n")}\n`, "utf8"); const nextSize = existing.byteLength + payload.byteLength; if (nextSize > (codec.maxBytes ?? DEFAULT_STRICT_JSONL_MAX_BYTES)) { @@ -167,5 +257,5 @@ export function appendStrictJsonlRecords( } else { createFileDurableExclusive(filePath, payload, trustedRoot); } - return readStrictJsonlSnapshot(filePath, codec); + return nextSize; } diff --git a/packages/artifacts/src/threat-model.ts b/packages/artifacts/src/threat-model.ts index ba2d4800d..b1a37033b 100644 --- a/packages/artifacts/src/threat-model.ts +++ b/packages/artifacts/src/threat-model.ts @@ -272,7 +272,6 @@ export const threatModelJsonSchema = { export type ThreatModel = z.infer; export type ThreatModelEvidenceReference = z.infer; -export type CapabilityStatus = (typeof CAPABILITY_STATUSES)[number]; export function validateThreatModel(value: unknown, path = "$"): SchemaValidationResult { return validateWithZod(threatModelSchema, value, { path, code: "THREAT_MODEL_SCHEMA_INVALID" }); diff --git a/packages/artifacts/src/usage-ledger.ts b/packages/artifacts/src/usage-ledger.ts index 034677421..fa59c125d 100644 --- a/packages/artifacts/src/usage-ledger.ts +++ b/packages/artifacts/src/usage-ledger.ts @@ -24,7 +24,6 @@ export const USAGE_FIELDS = [ "cache_write_tokens", "reasoning_tokens" ] as const; -export type UsageField = (typeof USAGE_FIELDS)[number]; export interface NormalizedUsage { model: string; diff --git a/packages/artifacts/src/workflow-contracts.ts b/packages/artifacts/src/workflow-contracts.ts index e9e2a93ee..b103ba3c6 100644 --- a/packages/artifacts/src/workflow-contracts.ts +++ b/packages/artifacts/src/workflow-contracts.ts @@ -34,7 +34,6 @@ import { import { canonicalTimestampSchema } from "./portable-json-primitives.js"; import { PROPERTY_PRIORITIES } from "./property-provenance.js"; import { SAFE_ID_PATTERN } from "./safe-paths.js"; -import { validateWithZod, type SchemaValidationResult } from "./schema-validation.js"; import { canonicalJsonValueKey } from "./lang-primitives.js"; const nonEmptyString = z.string().min(1); @@ -75,10 +74,6 @@ function withDocumentMetadata(schema: T, slug: string, vers }) as T; } -export const HARNESS_REPAIRS_SCHEMA_VERSION = "ultrafuzz.harness-repairs.v1" as const; -export const STRATEGY_DETECTIONS_SCHEMA_VERSION = "ultrafuzz.strategy-detections.v1" as const; -export const TRIAGED_FINDINGS_SCHEMA_VERSION = "ultrafuzz.triaged-findings.v1" as const; -export const SEVERITY_CLASSIFIED_FINDINGS_SCHEMA_VERSION = "ultrafuzz.severity-classified-findings.v1" as const; export const REFERENCE_MANIFEST_SCHEMA_VERSION = "ultrafuzz.reference-manifest.v1" as const; export const BOUNDARY_RECIPES_SCHEMA_VERSION = "ultrafuzz.boundary-recipes.v1" as const; export const ADMIN_CONFIG_BOUNDARY_MATRIX_SCHEMA_VERSION = "ultrafuzz.admin-config-boundary-matrix.v1" as const; @@ -1814,44 +1809,8 @@ export const findingLifecycleLedgerJsonSchema = workflowContractJsonSchemas["ult export const aggregationManifestJsonSchema = workflowContractJsonSchemas["ultrafuzz/aggregation-manifest@1"]; export const reportJsonSchema = workflowContractJsonSchemas["ultrafuzz/report@3"]; -export function validateWorkflowContract( - contract: WorkflowContractId, - value: unknown, - path = "$" -): SchemaValidationResult { - return validateWithZod(workflowContractSchemas[contract] as z.ZodType, value, { - path, - code: "WORKFLOW_ARTIFACT_SCHEMA_INVALID" - }); -} - -export type HarnessRepairs = z.infer; -export type StrategyDetections = z.infer; export type TriagedFindings = z.infer; export type SeverityClassifiedFindings = z.infer; -export type ReferenceManifest = z.infer; -export type BoundaryRecipes = z.infer; -export type AdminConfigBoundaryMatrix = z.infer; -export type DependencyScopeMatrix = z.infer; -export type ExternalizedStateAccounting = z.infer; -export type CoverageGoal = z.infer; -export type InvariantCampaignPlan = z.infer; -export type CampaignSummary = z.infer; -export type DifferentialPlan = z.infer; -export type ReferenceHarness = z.infer; -export type AuditedDifferentialLanes = z.infer; -export type DifferentialLaneResult = z.infer; -export type SemanticRedRegistry = z.infer; -export type DifferentialRedTriage = z.infer; -export type DifferentialRepairSummary = z.infer; -export type DifferentialGapReview = z.infer; -export type DifferentialReportReview = z.infer; -export type DynamicStrategyPlan = z.infer; -export type DynamicEnumeratorOutputs = z.infer; -export type SelectedStrategies = z.infer; -export type DynamicStrategyProvenance = z.infer; -export type FindingLifecycleLedger = z.infer; -export type AggregationManifest = z.infer; export type TerminalReport = z.infer; export const WORKFLOW_SCHEMA_FILES = { diff --git a/packages/artifacts/test/artifacts.test.ts b/packages/artifacts/test/artifacts.test.ts index 390b021cc..0387925e8 100644 --- a/packages/artifacts/test/artifacts.test.ts +++ b/packages/artifacts/test/artifacts.test.ts @@ -6,17 +6,13 @@ import path from "node:path"; import test from "node:test"; import { - ArtifactPathError, ArtifactSecretGateError, ARTIFACT_MANIFEST_FILE, - GENERATED_TESTS_SCHEMA_VERSION, - MAX_GENERATED_TEST_BUNDLE_BYTES, - MAX_GENERATED_TEST_BUNDLE_ENTRIES, appendUsageEvents, assertArtifactPublicationsContainNoSecrets, appendNodeAttempt, appendEvent, - appendLineDurable, + createEventRecord, assertUsageLedgerEntry, createRunLayout, getNodeArtifactDir, @@ -28,8 +24,6 @@ import { queryNodeAttempts, queryEvents, readArtifactManifest, - readEventQueryFacade, - readGeneratedTestManifest, readRunState, replayEvents, replayUsageEvents, @@ -42,7 +36,6 @@ import { verifyArtifactManifestPrerequisites, writeArtifact, writeArtifactManifest, - writeGeneratedTestManifest as writeGeneratedTestManifestWithFramework, writeRunState } from "../src/index.js"; @@ -50,18 +43,6 @@ function tempProject(): string { return fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-artifacts-")); } -function writeGeneratedTestManifest( - input: Omit[0], "framework"> -): ReturnType { - return writeGeneratedTestManifestWithFramework({ ...input, framework: "foundry" }); -} - -function errnoError(code: string): NodeJS.ErrnoException { - const error = new Error(`injected ${code}`) as NodeJS.ErrnoException; - error.code = code; - return error; -} - test("createRunLayout persists product-owned run evidence outside checkpoints", () => { const project = tempProject(); const layout = createRunLayout({ @@ -111,12 +92,11 @@ test("createRunLayout persists product-owned run evidence outside checkpoints", layout.statePath, layout.eventsPath, layout.usageLedgerPath, - layout.attemptLedgerPath, - path.join(layout.eventsIndexDir, "query-inputs.json") + layout.attemptLedgerPath ]) { assert.equal(fs.existsSync(expected), true, expected); } - for (const expected of [layout.artifactsDir, layout.reviewDir, layout.eventsIndexDir]) { + for (const expected of [layout.artifactsDir, layout.reviewDir]) { assert.equal(fs.statSync(expected).isDirectory(), true, expected); } assert.equal(path.basename(getNodeArtifactDir(layout, "node-a", { create: true })), "node-a"); @@ -146,7 +126,7 @@ test("validated usage events append idempotently with exact Smithers identities" assert.equal(ledger.entries[0]?.control_generation, "a".repeat(64)); assert.equal(ledger.entries.length, 1); - appendLineDurable(layout.usageLedgerPath, "{malformed", layout.root); + fs.appendFileSync(layout.usageLedgerPath, "{malformed\n"); assert.throws(() => replayUsageEvents(layout), /invalid strict JSON/u); }); @@ -191,11 +171,7 @@ test("usage ledger replay rejects entries copied from another run", () => { }; appendUsageEvents(firstLayout, [input]); appendUsageEvents(secondLayout, [input]); - appendLineDurable( - secondLayout.usageLedgerPath, - fs.readFileSync(firstLayout.usageLedgerPath, "utf8"), - secondLayout.root - ); + fs.appendFileSync(secondLayout.usageLedgerPath, fs.readFileSync(firstLayout.usageLedgerPath)); assert.throws(() => replayUsageEvents(secondLayout), /run_id belongs to/u); }); @@ -301,43 +277,33 @@ function failedAttemptInput(id: string, failureMessage: string) { }; } -test("node attempt ledger replay keeps a persisted failure redaction authoritative across credential rotation", () => { +test("node attempt ledger records an occurrence once and never re-derives it", () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: "attempt-ledger-redacted-replay" }); const oldCredential = "correct horse battery staple"; const input = failedAttemptInput("redacted-replay", `provider echoed ${oldCredential}`); const first = appendNodeAttempt(layout, { ...input, forbiddenSecretValues: [oldCredential] }); const persistedBytes = fs.readFileSync(layout.attemptLedgerPath); + assert.equal(first.appended, true); assert.equal(first.entry.failure_message, "provider echoed "); assert.deepEqual(first.entry.failure_message_redaction_span_code_points, [[...oldCredential].length]); assert.equal(first.entry.failure_message_truncated, undefined); - const replayed = appendNodeAttempt(layout, { - ...input, - forbiddenSecretValues: ["new credential value"] - }); - assert.equal(replayed.appended, false); - assert.deepEqual(replayed.entry, first.entry); - assert.deepEqual(fs.readFileSync(layout.attemptLedgerPath), persistedBytes); - - for (const failureMessage of [ - `different provider failure ${oldCredential}`, - `provider echoed ${oldCredential}; changed non-secret context` + // A replay of the same Smithers identity returns the stored entry whatever it + // would derive now: a rotated secret, a changed message, or a different outcome. + for (const replay of [ + { ...input, forbiddenSecretValues: ["new credential value"] }, + { ...input, failureMessage: `different provider failure ${oldCredential}` }, + { ...input, outcome: "canceled" as const, failureCategory: "canceled" as const, failureMessage: "run-cancelled" } ]) { - assert.throws( - () => - appendNodeAttempt(layout, { - ...input, - failureMessage, - forbiddenSecretValues: ["new credential value"] - }), - /already recorded with different immutable data/u - ); + const replayed = appendNodeAttempt(layout, replay); + assert.equal(replayed.appended, false); + assert.deepEqual(replayed.entry, first.entry); } assert.deepEqual(fs.readFileSync(layout.attemptLedgerPath), persistedBytes); }); -test("node attempt ledger uses explicit provenance for truncated redaction replay", () => { +test("node attempt ledger keeps truncation and redaction provenance for long failure messages", () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: "attempt-ledger-truncated-replay" }); const oldCredential = "old credential material ".repeat(30).trim(); const input = failedAttemptInput( @@ -346,42 +312,15 @@ test("node attempt ledger uses explicit provenance for truncated redaction repla ); const first = appendNodeAttempt(layout, { ...input, forbiddenSecretValues: [oldCredential] }); - const persistedBytes = fs.readFileSync(layout.attemptLedgerPath); assert.equal(Buffer.byteLength(first.entry.failure_message ?? "", "utf8"), MAX_NODE_ATTEMPT_FAILURE_MESSAGE_BYTES); assert.deepEqual(first.entry.failure_message_redaction_span_code_points, [[...oldCredential].length]); assert.equal(first.entry.failure_message_truncated, true); - const replayed = appendNodeAttempt(layout, { - ...input, - forbiddenSecretValues: ["rotated credential value"] + const literal = appendNodeAttempt(layout, { + ...failedAttemptInput("literal-placeholder", `provider echoed ; ${"x".repeat(1_500)}`) }); - assert.equal(replayed.appended, false); - assert.deepEqual(replayed.entry, first.entry); - assert.deepEqual(fs.readFileSync(layout.attemptLedgerPath), persistedBytes); - - assert.throws( - () => - appendNodeAttempt(layout, { - ...input, - failureMessage: `changed prefix ${oldCredential}; stable context ${"x".repeat(1_500)}`, - forbiddenSecretValues: ["rotated credential value"] - }), - /already recorded with different immutable data/u - ); -}); - -test("node attempt ledger does not infer replay-safe redaction from a literal placeholder", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "attempt-ledger-literal-placeholder" }); - const input = failedAttemptInput("literal-placeholder", `provider echoed ; ${"x".repeat(1_500)}`); - - const first = appendNodeAttempt(layout, input); - assert.equal(first.entry.failure_message_redaction_span_code_points, undefined); - assert.equal(first.entry.failure_message_truncated, true); - assert.throws( - () => - appendNodeAttempt(layout, { ...input, failureMessage: `provider echoed old credential; ${"x".repeat(1_500)}` }), - /already recorded with different immutable data/u - ); + assert.equal(literal.entry.failure_message_redaction_span_code_points, undefined); + assert.equal(literal.entry.failure_message_truncated, true); }); test("createRunLayout rejects symlinked run roots before creating outside writes", () => { @@ -796,7 +735,7 @@ test("artifact manifest reuse checks the complete prerequisite chain", () => { }); }); -test("events append to JSONL, replay, and expose query indexes", () => { +test("events append to JSONL, replay, and filter by node and status", () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-1" }); appendEvent(layout, { eventType: "node-synced", @@ -819,90 +758,53 @@ test("events append to JSONL, replay, and expose query indexes", () => { workflow_state: "in-progress" }); assert.equal(queryEvents(layout, { nodeId: "node-a", status: "succeeded" }).length, 1); - assert.equal(fs.existsSync(path.join(layout.eventsIndexDir, "node", "node-a.jsonl")), true); - assert.equal(fs.existsSync(path.join(layout.eventsIndexDir, "status", "succeeded.jsonl")), true); }); -test("event indexes encode long IDs in a collision-free hash namespace", () => { - const maximumRunId = "r".repeat(128); - const layout = createRunLayout({ projectRoot: tempProject(), runId: maximumRunId }); - const facadeBeforeAppend = readEventQueryFacade(layout); - const maximumNodeId = "n".repeat(128); - const directBoundaryNodeId = "d".repeat(122); - const longBoundaryNodeId = "e".repeat(123); - const legacyCollisionNodeId = `${maximumNodeId.slice(0, 97)}-${crypto - .createHash("sha256") - .update(maximumNodeId, "utf8") - .digest("hex") - .slice(0, 24)}`; - assert.equal(legacyCollisionNodeId.length, 122); - - const appendNodeSynced = (nodeId: string, workflowTaskId: string): void => { - appendEvent(layout, { - eventType: "node-synced", - nodeId, - status: "running", - payload: { workflow_run_id: "workflow-1", workflow_task_id: workflowTaskId } - }); - }; - - appendNodeSynced(maximumNodeId, "task-maximum"); - appendNodeSynced(longBoundaryNodeId, "task-long-boundary"); - appendNodeSynced(legacyCollisionNodeId, "task-legacy-collision"); - appendNodeSynced(directBoundaryNodeId, "task-direct-boundary"); - - const hashedIndexPath = (dimension: string, value: string): string => - path.join( - layout.eventsIndexDir, - dimension, - "sha256", - `${crypto.createHash("sha256").update(value, "utf8").digest("hex")}.jsonl` - ); - const maximumRunIndex = hashedIndexPath("run", maximumRunId); - assert.equal(fs.existsSync(maximumRunIndex), true); - assert.equal(fs.existsSync(path.join(layout.eventsIndexDir, "node", `${legacyCollisionNodeId}.jsonl`)), true); - assert.equal(fs.existsSync(path.join(layout.eventsIndexDir, "node", `${directBoundaryNodeId}.jsonl`)), true); - assert.equal(fs.existsSync(hashedIndexPath("node", longBoundaryNodeId)), true); - assert.equal(fs.existsSync(hashedIndexPath("node", maximumNodeId)), true); - - const maximumRecords = fs - .readFileSync(maximumRunIndex, "utf8") - .trimEnd() - .split("\n") - .map((line) => JSON.parse(line) as { run_id: string }); - assert.equal(maximumRecords.length, 4); - assert.equal( - maximumRecords.every((record) => record.run_id === maximumRunId), - true - ); - const facadeAfterAppend = readEventQueryFacade(layout) as { - filters?: unknown; - long_filters?: unknown; - index_key_encoding?: unknown; - }; - assert.deepEqual(facadeAfterAppend, facadeBeforeAppend); - assert.deepEqual(facadeAfterAppend.filters, { - run_id: "events.index/run/.jsonl", - node_id: "events.index/node/.jsonl", - event_type: "events.index/type/.jsonl", - status: "events.index/status/.jsonl", - timestamp: "events.index/timestamp/.jsonl" - }); - assert.deepEqual(facadeAfterAppend.long_filters, { - run_id: "events.index/run/sha256/.jsonl", - node_id: "events.index/node/sha256/.jsonl", - event_type: "events.index/type/sha256/.jsonl", - status: "events.index/status/sha256/.jsonl" +test("event appends and replays keep working past 100,000 records", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-long-journal" }); + const template = createEventRecord(layout, { + eventType: "node-synced", + nodeId: "node-a", + status: "running", + timestamp: "2026-08-05T00:00:00.000Z", + payload: { workflow_run_id: "workflow-1", workflow_task_id: "task-1" } }); - assert.deepEqual(facadeAfterAppend.index_key_encoding, { - version: "ultrafuzz.event-index-key.v1", - direct_max_id_length: 122, - direct_id_path: "/.jsonl", - long_id_path: "/sha256/.jsonl", - digest: "sha256", - hash_input_encoding: "utf8", - digest_encoding: "hex" + const lines = Array.from({ length: 100_000 }, (_, index) => + JSON.stringify({ ...template, event_id: `evt-${index.toString(16).padStart(24, "0")}` }) + ); + fs.writeFileSync(layout.eventsPath, `${lines.join("\n")}\n`); + + const appended = appendEvent(layout, { + eventType: "node-synced", + nodeId: "node-a", + status: "succeeded", + timestamp: "2026-08-05T00:00:01.000Z", + payload: { workflow_run_id: "workflow-1", workflow_task_id: "task-1" } }); + const replay = replayEvents(layout, Number.MAX_SAFE_INTEGER); + assert.equal(replay.records.length, 100_001); + assert.equal(replay.records.at(-1)?.event_id, appended.event_id); +}); + +test("an event append refuses a repeated or out-of-order event without changing the journal", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-event-window" }); + const event = (nodeId: string, timestamp: string) => + ({ + eventType: "node-synced", + nodeId, + status: "succeeded", + timestamp, + payload: { workflow_run_id: "workflow-1", workflow_task_id: `task-${nodeId}` } + }) as const; + appendEvent(layout, event("node-a", "2026-08-05T00:00:01.000Z")); + appendEvent(layout, event("node-b", "2026-08-05T00:00:01.000Z")); + const before = fs.readFileSync(layout.eventsPath); + + // The same event again in the same millisecond, behind a different one, is still a repeat. + assert.throws(() => appendEvent(layout, event("node-a", "2026-08-05T00:00:01.000Z")), /duplicate identity/u); + assert.throws(() => appendEvent(layout, event("node-c", "2026-08-05T00:00:00.000Z")), /not ordered/u); + assert.deepEqual(fs.readFileSync(layout.eventsPath), before); + assert.equal(replayEvents(layout).records.length, 2); }); test("event redaction covers token families, AWS keys, URL credentials, and private keys", () => { @@ -1118,6 +1020,33 @@ test("canonical publication secret gate fails closed without rewriting bytes", ( ) ); + // The positive rules must not fire on ordinary contract output either: a + // Foundry test deriving and signing with Anvil's published mnemonic and + // account (0) key, and a qualified identifier with three long dotted + // segments (JWT-shaped until the rule required an eyJ header). + assert.doesNotThrow(() => + assertArtifactPublicationsContainNoSecrets( + new Map([ + [ + "generated-tests/AnvilSigner.t.sol", + Buffer.from( + 'string memory mnemonic = "test test test test test test test test test test test junk";\n' + + "uint256 privateKey = 0xac0974bec39a17e36ba4a6b4d238ff944bacb478cbed5efcae784d7bf4f2ff80;\n" + // gitleaks:allow -- public Anvil dev key fixture for the redaction tests + "assertEq(vm.deriveKey(mnemonic, 0), privateKey);\n", + "utf8" + ) + ], + [ + "setup/call-graph.md", + Buffer.from( + "`ReentrancyGuardUpgradeable.nonReentrantModifier.lockedStateCheck` reverts on re-entry\n", + "utf8" + ) + ] + ]) + ) + ); + // Every credential format the previous hand-rolled patterns could name is // still rejected — via a secretlint library finding or a documented // supplemental pattern. @@ -1181,649 +1110,3 @@ test("updateNodeState accepts an explicit transition timestamp", () => { assert.equal(running.last_transition_at, "2026-01-01T00:00:05.000Z"); assert.equal(running.nodes["node-a"]?.wait_since, "2026-01-01T00:00:05.000Z"); }); - -test("durable append rejects a symlinked parent before creating outside directories", () => { - const root = tempProject(); - const outside = tempProject(); - fs.symlinkSync(outside, path.join(root, "linked"), "dir"); - - assert.throws( - () => appendLineDurable(path.join(root, "linked", "created", "audit.jsonl"), "entry", root), - /crosses symlink/u - ); - assert.equal(fs.existsSync(path.join(outside, "created")), false); -}); - -test("durable append rejects a half write without retrying the record", (t) => { - const root = tempProject(); - const filePath = path.join(root, "audit.jsonl"); - const expected = Buffer.from("0123456789\n", "utf8"); - const partialLength = Math.floor(expected.length / 2); - const realWriteSync = fs.writeSync; - const realCloseSync = fs.closeSync; - const writeSync = t.mock.method(fs, "writeSync", (fd: number, bytes: Uint8Array) => - realWriteSync(fd, Buffer.from(bytes).subarray(0, partialLength)) - ); - const closeSync = t.mock.method(fs, "closeSync", (fd: number) => realCloseSync(fd)); - - assert.throws( - () => appendLineDurable(filePath, "0123456789"), - (error: unknown) => { - assert.ok(error instanceof ArtifactPathError); - assert.equal(error.code, "short-write"); - assert.match(error.message, new RegExp(`wrote ${partialLength} of ${expected.length} bytes`, "u")); - return true; - } - ); - - assert.equal(writeSync.mock.callCount(), 1); - assert.equal(closeSync.mock.callCount(), 1); - assert.deepEqual(fs.readFileSync(filePath), expected.subarray(0, partialLength)); -}); - -test("durable append rejects a zero write without retrying the record", (t) => { - const root = tempProject(); - const filePath = path.join(root, "audit.jsonl"); - const realCloseSync = fs.closeSync; - const writeSync = t.mock.method(fs, "writeSync", () => 0); - const closeSync = t.mock.method(fs, "closeSync", (fd: number) => realCloseSync(fd)); - - assert.throws( - () => appendLineDurable(filePath, "entry"), - (error: unknown) => { - assert.ok(error instanceof ArtifactPathError); - assert.equal(error.code, "short-write"); - assert.match(error.message, /wrote 0 of 6 bytes/u); - return true; - } - ); - - assert.equal(writeSync.mock.callCount(), 1); - assert.equal(closeSync.mock.callCount(), 1); - assert.equal(fs.readFileSync(filePath, "utf8"), ""); -}); - -test("durable append propagates a directory fsync I/O failure", (t) => { - const root = tempProject(); - const filePath = path.join(root, "audit.jsonl"); - const realFsyncSync = fs.fsyncSync; - const realCloseSync = fs.closeSync; - let fsyncCall = 0; - const fsyncSync = t.mock.method(fs, "fsyncSync", (fd: number) => { - fsyncCall += 1; - if (fsyncCall === 1) { - realFsyncSync(fd); - return; - } - throw errnoError("EIO"); - }); - const closeSync = t.mock.method(fs, "closeSync", (fd: number) => realCloseSync(fd)); - - assert.throws( - () => appendLineDurable(filePath, "entry"), - (error: unknown) => { - assert.ok(error instanceof Error && "code" in error); - assert.equal(error.code, "EIO"); - return true; - } - ); - - assert.equal(fsyncSync.mock.callCount(), 2); - assert.equal(closeSync.mock.callCount(), 2); - assert.equal(fs.readFileSync(filePath, "utf8"), "entry\n"); -}); - -test("durable append tolerates only an unsupported directory fsync operation", (t) => { - const root = tempProject(); - const filePath = path.join(root, "audit.jsonl"); - const realFsyncSync = fs.fsyncSync; - const realCloseSync = fs.closeSync; - let fsyncCall = 0; - const fsyncSync = t.mock.method(fs, "fsyncSync", (fd: number) => { - fsyncCall += 1; - if (fsyncCall === 1) { - realFsyncSync(fd); - return; - } - throw errnoError("EINVAL"); - }); - const closeSync = t.mock.method(fs, "closeSync", (fd: number) => realCloseSync(fd)); - - appendLineDurable(filePath, "entry"); - - assert.equal(fsyncSync.mock.callCount(), 2); - assert.equal(closeSync.mock.callCount(), 2); - assert.equal(fs.readFileSync(filePath, "utf8"), "entry\n"); -}); - -test("durable append propagates a close failure without retrying close", (t) => { - const root = tempProject(); - const filePath = path.join(root, "audit.jsonl"); - const realCloseSync = fs.closeSync; - const closeSync = t.mock.method(fs, "closeSync", (fd: number) => { - realCloseSync(fd); - throw errnoError("EIO"); - }); - - assert.throws( - () => appendLineDurable(filePath, "entry"), - (error: unknown) => { - assert.ok(error instanceof Error && "code" in error); - assert.equal(error.code, "EIO"); - return true; - } - ); - - assert.equal(closeSync.mock.callCount(), 1); - assert.equal(fs.readFileSync(filePath, "utf8"), "entry\n"); -}); - -test("durable append aggregates an operation failure with its close failure", (t) => { - const root = tempProject(); - const filePath = path.join(root, "audit.jsonl"); - const realCloseSync = fs.closeSync; - const writeSync = t.mock.method(fs, "writeSync", () => { - throw errnoError("EIO"); - }); - const closeSync = t.mock.method(fs, "closeSync", (fd: number) => { - realCloseSync(fd); - throw errnoError("EBADF"); - }); - - assert.throws( - () => appendLineDurable(filePath, "entry"), - (error: unknown) => { - assert.ok(error instanceof AggregateError); - assert.deepEqual( - error.errors.map((entry: unknown) => (entry instanceof Error && "code" in entry ? entry.code : undefined)), - ["EIO", "EBADF"] - ); - return true; - } - ); - - assert.equal(writeSync.mock.callCount(), 1); - assert.equal(closeSync.mock.callCount(), 1); - assert.equal(fs.readFileSync(filePath, "utf8"), ""); -}); - -test("generated-test manifests persist explicit generated files with provenance", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-1" }); - const generatedContents = "contract InvariantTest {} // π\n"; - const supportContents = "library InvariantFixture {} // café\n"; - const manifest = writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - provenance: { agent_ref: "CodexAgent", workflow_task_id: "node:strategy-a", attempt_index: 0 }, - tests: [ - { - path: "generated-tests/Invariant.t.sol", - content: generatedContents, - language: "solidity" - } - ], - supportFiles: [ - { - path: "generated-tests/helpers/InvariantFixture.sol", - content: supportContents, - language: "solidity" - } - ] - }); - - assert.equal(manifest.schema_version, GENERATED_TESTS_SCHEMA_VERSION); - assert.equal(manifest.framework, "foundry"); - assert.equal(manifest.generated_tests.length, 1); - assert.equal(manifest.generated_tests[0]!.path, "generated-tests/Invariant.t.sol"); - assert.equal(manifest.generated_tests[0]!.size_bytes, Buffer.byteLength(generatedContents)); - assert.equal( - manifest.generated_tests[0]!.sha256, - crypto.createHash("sha256").update(generatedContents).digest("hex") - ); - assert.equal(manifest.generated_tests[0]!.provenance!.agent_ref, "CodexAgent"); - assert.equal(manifest.support_files[0]!.path, "generated-tests/helpers/InvariantFixture.sol"); - assert.equal(manifest.support_files[0]!.size_bytes, Buffer.byteLength(supportContents)); - assert.equal(manifest.support_files[0]!.sha256, crypto.createHash("sha256").update(supportContents).digest("hex")); - assert.equal( - fs.existsSync(path.join(getNodeArtifactDir(layout, "strategy-a"), "generated-tests", "Invariant.t.sol")), - true - ); -}); - -test("generated-test manifest writer rejects zero-byte companion files", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-empty-generated-test" }); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - supportFiles: [], - tests: [{ path: "generated-tests/Empty.t.sol", content: "" }] - }), - /generated test file must be non-empty/u - ); - assert.equal(fs.existsSync(path.join(getNodeArtifactDir(layout, "strategy-a"), "generated-tests.json")), false); -}); - -test("generated-test manifest writer rejects support-only bundles before writing files", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-support-only-generated-test" }); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [], - supportFiles: [{ path: "generated-tests/Helper.sol", content: "library Helper {}\n" }] - }), - /cannot declare support files without a runnable generated test/u - ); - assert.equal(fs.existsSync(path.join(getNodeArtifactDir(layout, "strategy-a"), "generated-tests.json")), false); - assert.equal( - fs.existsSync(path.join(getNodeArtifactDir(layout, "strategy-a"), "generated-tests", "Helper.sol")), - false - ); -}); - -test("generated-test manifest writer rejects excess combined entries before creating bundle paths", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-excess-generated-test-entries" }); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: Array.from({ length: MAX_GENERATED_TEST_BUNDLE_ENTRIES + 1 }, (_, index) => ({ - path: `generated-tests/Test-${index}.sol`, - content: "x" - })), - supportFiles: [] - }), - /1024-entry combined bundle limit/u - ); - assert.equal(fs.existsSync(path.join(layout.artifactsDir, "strategy-a")), false); -}); - -test("generated-test manifest writer rejects excess cumulative existing bytes before reading companions", (t) => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-excess-generated-test-bytes" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - fs.mkdirSync(generatedTestsDir); - const tests = Array.from({ length: 5 }, (_, index) => { - const relativePath = `generated-tests/Test-${index}.sol`; - const absolutePath = path.join(nodeDir, relativePath); - fs.writeFileSync(absolutePath, "x", "utf8"); - fs.truncateSync(absolutePath, MAX_GENERATED_TEST_BUNDLE_BYTES / 4); - return { path: relativePath }; - }); - const readSync = t.mock.method(fs, "readSync", () => { - throw new Error("companion content was read before cumulative resource preflight completed"); - }); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests, - supportFiles: [] - }), - /67108864-byte combined companion limit/u - ); - assert.equal(readSync.mock.callCount(), 0); - assert.equal(fs.existsSync(path.join(nodeDir, "generated-tests.json")), false); -}); - -test("generated-test manifest writer rejects cross-array duplicate paths before changing bundle bytes", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-duplicate-generated-test" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - fs.mkdirSync(generatedTestsDir, { recursive: true }); - const companionPath = path.join(generatedTestsDir, "Shared.sol"); - const manifestPath = path.join(nodeDir, "generated-tests.json"); - fs.writeFileSync(companionPath, "sentinel companion\n", "utf8"); - fs.writeFileSync(manifestPath, "sentinel manifest\n", "utf8"); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Shared.sol", content: "replacement test\n" }], - supportFiles: [{ path: "generated-tests/Shared.sol", content: "replacement support\n" }] - }), - /repeats path "generated-tests\/Shared\.sol"/u - ); - assert.equal(fs.readFileSync(companionPath, "utf8"), "sentinel companion\n"); - assert.equal(fs.readFileSync(manifestPath, "utf8"), "sentinel manifest\n"); -}); - -test("generated-test manifest writer preflights missing existing companions before overwriting earlier files", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-missing-support-generated-test" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - fs.mkdirSync(generatedTestsDir, { recursive: true }); - const testPath = path.join(generatedTestsDir, "Replay.t.sol"); - const manifestPath = path.join(nodeDir, "generated-tests.json"); - fs.writeFileSync(testPath, "sentinel test\n", "utf8"); - fs.writeFileSync(manifestPath, "sentinel manifest\n", "utf8"); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Replay.t.sol", content: "replacement test\n" }], - supportFiles: [{ path: "generated-tests/MissingHelper.sol" }] - }), - /generated-test support file does not exist/u - ); - assert.equal(fs.readFileSync(testPath, "utf8"), "sentinel test\n"); - assert.equal(fs.readFileSync(manifestPath, "utf8"), "sentinel manifest\n"); -}); - -test("generated-test manifest writer preflights non-file destinations before overwriting earlier files", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-directory-support-generated-test" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - const supportDirectory = path.join(generatedTestsDir, "InvariantFixture.sol"); - fs.mkdirSync(supportDirectory, { recursive: true }); - const testPath = path.join(generatedTestsDir, "Replay.t.sol"); - const manifestPath = path.join(nodeDir, "generated-tests.json"); - fs.writeFileSync(testPath, "sentinel test\n", "utf8"); - fs.writeFileSync(manifestPath, "sentinel manifest\n", "utf8"); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Replay.t.sol", content: "replacement test\n" }], - supportFiles: [{ path: "generated-tests/InvariantFixture.sol", content: "library InvariantFixture {}\n" }] - }), - /generated-test bundle destination must be a regular file/u - ); - assert.equal(fs.readFileSync(testPath, "utf8"), "sentinel test\n"); - assert.equal(fs.readFileSync(manifestPath, "utf8"), "sentinel manifest\n"); - assert.deepEqual(fs.readdirSync(supportDirectory), []); -}); - -test("generated-test manifest writer validates entry metadata before changing bundle bytes", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-invalid-metadata-generated-test" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - fs.mkdirSync(generatedTestsDir, { recursive: true }); - const testPath = path.join(generatedTestsDir, "Replay.t.sol"); - const manifestPath = path.join(nodeDir, "generated-tests.json"); - fs.writeFileSync(testPath, "sentinel test\n", "utf8"); - fs.writeFileSync(manifestPath, "sentinel manifest\n", "utf8"); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Replay.t.sol", content: "replacement test\n" }], - supportFiles: [ - { - path: "generated-tests/InvariantFixture.sol", - content: "library InvariantFixture {}\n", - language: "" - } - ] - }), - /generated tests manifest is schema-invalid/u - ); - assert.equal(fs.readFileSync(testPath, "utf8"), "sentinel test\n"); - assert.equal(fs.existsSync(path.join(generatedTestsDir, "InvariantFixture.sol")), false); - assert.equal(fs.readFileSync(manifestPath, "utf8"), "sentinel manifest\n"); -}); - -test("generated-test manifest writer preflights its manifest destination before changing companion bytes", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-directory-manifest-generated-test" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - const manifestDirectory = path.join(nodeDir, "generated-tests.json"); - fs.mkdirSync(generatedTestsDir, { recursive: true }); - fs.mkdirSync(manifestDirectory); - const testPath = path.join(generatedTestsDir, "Replay.t.sol"); - fs.writeFileSync(testPath, "sentinel test\n", "utf8"); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Replay.t.sol", content: "replacement test\n" }], - supportFiles: [] - }), - /generated-test bundle destination must be a regular file/u - ); - assert.equal(fs.readFileSync(testPath, "utf8"), "sentinel test\n"); - assert.deepEqual(fs.readdirSync(manifestDirectory), []); -}); - -test("generated-test manifest writer rejects a hard-linked manifest destination before changing bundle bytes", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-hardlinked-manifest-generated-test" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - fs.mkdirSync(generatedTestsDir, { recursive: true }); - const testPath = path.join(generatedTestsDir, "Replay.t.sol"); - const manifestPath = path.join(nodeDir, "generated-tests.json"); - const manifestAlias = path.join(tempProject(), "generated-tests-alias.json"); - fs.writeFileSync(testPath, "sentinel test\n", "utf8"); - fs.writeFileSync(manifestPath, "sentinel manifest\n", "utf8"); - fs.linkSync(manifestPath, manifestAlias); - const manifestInode = fs.lstatSync(manifestPath).ino; - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Replay.t.sol", content: "replacement test\n" }], - supportFiles: [] - }), - (error: unknown) => { - assert.ok(error instanceof ArtifactPathError); - assert.equal(error.code, "hard-link"); - assert.match(error.message, /bundle destination must be singly linked/u); - return true; - } - ); - assert.equal(fs.readFileSync(testPath, "utf8"), "sentinel test\n"); - assert.equal(fs.readFileSync(manifestPath, "utf8"), "sentinel manifest\n"); - assert.equal(fs.readFileSync(manifestAlias, "utf8"), "sentinel manifest\n"); - assert.equal(fs.lstatSync(manifestPath).ino, manifestInode); - assert.equal(fs.lstatSync(manifestPath).nlink, 2); -}); - -test("generated-test manifest writer rejects a hard-linked companion before changing bundle bytes", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-hardlinked-companion-generated-test" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - fs.mkdirSync(generatedTestsDir, { recursive: true }); - const testPath = path.join(generatedTestsDir, "Replay.t.sol"); - const supportPath = path.join(generatedTestsDir, "InvariantFixture.sol"); - const supportAlias = path.join(tempProject(), "InvariantFixture-alias.sol"); - const manifestPath = path.join(nodeDir, "generated-tests.json"); - fs.writeFileSync(testPath, "sentinel test\n", "utf8"); - fs.writeFileSync(supportPath, "sentinel support\n", "utf8"); - fs.linkSync(supportPath, supportAlias); - fs.writeFileSync(manifestPath, "sentinel manifest\n", "utf8"); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Replay.t.sol", content: "replacement test\n" }], - supportFiles: [{ path: "generated-tests/InvariantFixture.sol", content: "replacement support\n" }] - }), - (error: unknown) => { - assert.ok(error instanceof ArtifactPathError); - assert.equal(error.code, "hard-link"); - assert.match(error.message, /bundle destination must be singly linked/u); - return true; - } - ); - assert.equal(fs.readFileSync(testPath, "utf8"), "sentinel test\n"); - assert.equal(fs.readFileSync(supportPath, "utf8"), "sentinel support\n"); - assert.equal(fs.readFileSync(supportAlias, "utf8"), "sentinel support\n"); - assert.equal(fs.readFileSync(manifestPath, "utf8"), "sentinel manifest\n"); - assert.equal(fs.lstatSync(supportPath).nlink, 2); -}); - -test("generated-test manifest reader rejects a hard-linked manifest", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-read-hardlinked-generated-test" }); - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Replay.t.sol", content: "contract Replay {}\n" }], - supportFiles: [] - }); - const manifestPath = path.join(getNodeArtifactDir(layout, "strategy-a"), "generated-tests.json"); - const manifestAlias = path.join(tempProject(), "generated-tests-alias.json"); - fs.linkSync(manifestPath, manifestAlias); - const manifestBytes = fs.readFileSync(manifestPath); - - assert.throws( - () => readGeneratedTestManifest(layout, "strategy-a"), - (error: unknown) => { - assert.ok(error instanceof ArtifactPathError); - assert.equal(error.code, "hard-link"); - assert.match(error.message, /manifest must be a singly linked regular file/u); - return true; - } - ); - assert.deepEqual(fs.readFileSync(manifestPath), manifestBytes); - assert.deepEqual(fs.readFileSync(manifestAlias), manifestBytes); -}); - -test("generated-test manifest writer rejects file-directory path collisions before creating bundle bytes", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-prefix-collision-generated-test" }); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Replay.t.sol", content: "contract Replay {}\n" }], - supportFiles: [ - { - path: "generated-tests/Replay.t.sol/InvariantFixture.sol", - content: "library InvariantFixture {}\n" - } - ] - }), - /conflicts with file path/u - ); - assert.equal(fs.existsSync(path.join(layout.artifactsDir, "strategy-a")), false); -}); - -test("generated-test manifest writer rejects noncanonical path aliases without rewriting them", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-noncanonical-path-generated-test" }); - - for (const candidate of [ - "Replay.t.sol", - "generated-tests/sub/../Replay.t.sol", - "generated-tests/./Replay.t.sol", - "generated-tests/sub//Replay.t.sol" - ]) { - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: candidate, content: "contract Replay {}\n" }], - supportFiles: [] - }), - /must (?:begin with|already be a normalized relative POSIX path)/u, - candidate - ); - assert.equal(fs.existsSync(path.join(layout.artifactsDir, "strategy-a")), false, candidate); - } -}); - -test("generated-test manifest writer rejects non-UTF-8 supplied support before creating bundle bytes", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-binary-support-generated-test" }); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - tests: [{ path: "generated-tests/Replay.t.sol", content: "contract Replay {}\n" }], - supportFiles: [{ path: "generated-tests/fixture.dat", content: Buffer.from([0xff]) }] - }), - /generated-test support file must be strict UTF-8 text/u - ); - assert.equal(fs.existsSync(path.join(getNodeArtifactDir(layout, "strategy-a"), "generated-tests.json")), false); - assert.equal(fs.existsSync(path.join(getNodeArtifactDir(layout, "strategy-a"), "generated-tests")), false); -}); - -test("generated-test manifest writer rejects final symlinks without touching outside files", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-symlinked-generated-test" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - fs.mkdirSync(generatedTestsDir, { recursive: true }); - const outsideDir = tempProject(); - const outsideFile = path.join(outsideDir, "Outside.t.sol"); - fs.writeFileSync(outsideFile, "outside sentinel\n"); - const symlinkPath = path.join(generatedTestsDir, "Linked.t.sol"); - fs.symlinkSync(outsideFile, symlinkPath); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - supportFiles: [], - tests: [{ path: "generated-tests/Linked.t.sol", content: "replacement\n" }] - }), - /symlink/u - ); - assert.equal(fs.lstatSync(symlinkPath).isSymbolicLink(), true); - assert.equal(fs.readFileSync(outsideFile, "utf8"), "outside sentinel\n"); - assert.equal(fs.existsSync(path.join(nodeDir, "generated-tests.json")), false); -}); - -test("generated-test manifest writer rejects broken final symlinks before writing content", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-broken-symlink-generated-test" }); - const nodeDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); - const generatedTestsDir = path.join(nodeDir, "generated-tests"); - fs.mkdirSync(generatedTestsDir, { recursive: true }); - const missingOutsideFile = path.join(tempProject(), "Missing.t.sol"); - const symlinkPath = path.join(generatedTestsDir, "Broken.t.sol"); - fs.symlinkSync(missingOutsideFile, symlinkPath); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - supportFiles: [], - tests: [{ path: "generated-tests/Broken.t.sol", content: "replacement\n" }] - }), - /symlink/u - ); - assert.equal(fs.lstatSync(symlinkPath).isSymbolicLink(), true); - assert.equal(fs.existsSync(missingOutsideFile), false); -}); - -test("generated-test manifest writer rejects paths outside the generated-tests root", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-outside-generated-test" }); - - assert.throws( - () => - writeGeneratedTestManifest({ - layout, - nodeId: "strategy-a", - supportFiles: [], - tests: [{ path: "generated-tests/../../Outside.t.sol", content: "outside\n" }] - }), - /cannot traverse outside/u - ); - assert.equal(fs.existsSync(path.join(layout.artifactsDir, "Outside.t.sol")), false); -}); diff --git a/packages/artifacts/test/events-tail-repair.test.ts b/packages/artifacts/test/events-tail-repair.test.ts index e4565cdde..b0a38abdb 100644 --- a/packages/artifacts/test/events-tail-repair.test.ts +++ b/packages/artifacts/test/events-tail-repair.test.ts @@ -5,15 +5,7 @@ import os from "node:os"; import path from "node:path"; import test from "node:test"; -import { - appendBytesDurableAt, - appendEvent, - appendEventRecord, - createEventRecord, - createRunLayout, - replayEvents, - truncateDurable -} from "../src/index.js"; +import { appendBytesDurableAt, appendEvent, createEventRecord, createRunLayout, replayEvents } from "../src/index.js"; function tempProject(): string { return fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-events-tail-")); @@ -29,16 +21,19 @@ test("a torn event journal is rejected without repair or append", () => { payload: { workflow_run_id: "workflow-1", workflow_task_id: "task-1", attempt: 1 } }); fs.writeFileSync(layout.eventsPath, `${JSON.stringify(first)}\n{"event_id":"evt-tor`, "utf8"); - const second = createEventRecord(layout, { - eventType: "node-synced", - nodeId: "node-b", - status: "succeeded", - timestamp: "2026-08-05T00:00:01.000Z", - payload: { workflow_run_id: "workflow-1", workflow_task_id: "task-2", attempt: 2 } - }); const before = fs.readFileSync(layout.eventsPath); - assert.throws(() => appendEventRecord(layout.eventsPath, second), /torn or unterminated/u); + assert.throws( + () => + appendEvent(layout, { + eventType: "node-synced", + nodeId: "node-b", + status: "succeeded", + timestamp: "2026-08-05T00:00:01.000Z", + payload: { workflow_run_id: "workflow-1", workflow_task_id: "task-2", attempt: 2 } + }), + /torn or unterminated/u + ); assert.deepEqual(fs.readFileSync(layout.eventsPath), before); assert.throws(() => replayEvents(layout), /torn or unterminated/u); }); @@ -52,33 +47,9 @@ test("a complete but unterminated trailing object is rejected without repair", ( timestamp: "2026-08-05T00:00:00.000Z", payload: { workflow_run_id: "workflow-1", workflow_task_id: "task-1" } }); - const second = createEventRecord(layout, { - eventType: "node-synced", - nodeId: "node-b", - status: "succeeded", - timestamp: "2026-08-05T00:00:01.000Z", - payload: { workflow_run_id: "workflow-1", workflow_task_id: "task-2" } - }); fs.writeFileSync(layout.eventsPath, JSON.stringify(first), "utf8"); const before = fs.readFileSync(layout.eventsPath); - assert.throws(() => appendEventRecord(layout.eventsPath, second), /torn or unterminated/u); - assert.deepEqual(fs.readFileSync(layout.eventsPath), before); -}); - -test("a torn event index rejects the whole append before the canonical journal changes", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-index-torn" }); - const first = appendEvent(layout, { - eventType: "node-synced", - nodeId: "node-a", - status: "succeeded", - timestamp: "2026-08-05T00:00:00.000Z", - payload: { workflow_run_id: "workflow-1", workflow_task_id: "task-1" } - }); - const indexPath = path.join(layout.eventsIndexDir, "run", `${layout.runId}.jsonl`); - fs.appendFileSync(indexPath, '{"event_id":"evt-tor', "utf8"); - - const canonicalBefore = fs.readFileSync(layout.eventsPath); assert.throws( () => appendEvent(layout, { @@ -90,17 +61,15 @@ test("a torn event index rejects the whole append before the canonical journal c }), /torn or unterminated/u ); - assert.deepEqual(fs.readFileSync(layout.eventsPath), canonicalBefore); - assert.equal(replayEvents(layout).records[0]?.event_id, first.event_id); + assert.deepEqual(fs.readFileSync(layout.eventsPath), before); }); -test("durable repair mutations reject stale sizes and hard-linked files", () => { +test("durable appends reject stale sizes and hard-linked files", () => { const filePath = path.join(tempProject(), "events.jsonl"); fs.writeFileSync(filePath, "first\nsecond", "utf8"); const observedSize = fs.statSync(filePath).size; fs.appendFileSync(filePath, "-raced\n", "utf8"); - assert.throws(() => truncateDurable(filePath, 6, { expectedSize: observedSize }), /changed size/u); assert.throws( () => appendBytesDurableAt(filePath, Buffer.from("\n"), { expectedSize: observedSize }), /changed size/u @@ -109,7 +78,6 @@ test("durable repair mutations reject stale sizes and hard-linked files", () => const currentSize = fs.statSync(filePath).size; const linkPath = path.join(path.dirname(filePath), "events-link.jsonl"); fs.linkSync(filePath, linkPath); - assert.throws(() => truncateDurable(filePath, 0, { expectedSize: currentSize }), /must not be hard-linked/u); assert.throws( () => appendBytesDurableAt(filePath, Buffer.from("x"), { expectedSize: currentSize }), /must not be hard-linked/u diff --git a/packages/artifacts/test/json-validator-preflight.test.ts b/packages/artifacts/test/json-validator-preflight.test.ts index 9401182a7..7555f0351 100644 --- a/packages/artifacts/test/json-validator-preflight.test.ts +++ b/packages/artifacts/test/json-validator-preflight.test.ts @@ -63,7 +63,6 @@ function identityGate(document: unknown) { schemaId: binding.schema_id, schemaSha256: binding.schema_sha256, schemaBundleSha256: binding.schema_bundle_sha256, - validatorBuild: binding.validator_build, artifactSha256: ARTIFACT_VALIDATOR_SMOKE_FIXTURE_SHA256 } } @@ -86,6 +85,19 @@ test("validator preflight parser accepts only the exact non-transforming success ); }); +test("validator preflight parser accepts the same schemas reported by another validator build", () => { + // #921: a rebuild that only changes the validator modules must not fail an in-flight run's + // preflight, whose trusted CLI was sealed by the earlier build. The schema identity still binds. + const value = successEnvelope(); + objectField(objectField(value, "data"), "schema").validator_build = `ultrafuzz-json-validator.v1:${"9".repeat(64)}`; + + assert.deepEqual(parseJsonValidatorPreflightSuccessEnvelope(encode(value)), value); + assert.equal(identityGate(value)?.status, "passed"); + objectField(objectField(value, "data"), "schema").sha256 = "0".repeat(64); + assert.throws(() => parseJsonValidatorPreflightSuccessEnvelope(encode(value)), /mismatched identity/u); + assert.equal(identityGate(value)?.status, "failed"); +}); + test("validator preflight parser can authenticate a sealed historical identity explicitly", () => { const value = successEnvelope(); const schema = objectField(objectField(value, "data"), "schema"); @@ -97,7 +109,6 @@ test("validator preflight parser can authenticate a sealed historical identity e schemaId: binding.schema_id, schemaSha256: binding.schema_sha256, schemaBundleSha256: schema.bundle_sha256 as string, - validatorBuild: schema.validator_build as string, artifactSha256: ARTIFACT_VALIDATOR_SMOKE_FIXTURE_SHA256 }; @@ -172,11 +183,6 @@ const contractMutations: ReadonlyArray<{ name: "missing validator build", mutate: (value) => void delete objectField(objectField(value, "data"), "schema").validator_build }, - { - name: "wrong validator build", - structurallyValid: true, - mutate: (value) => void (objectField(objectField(value, "data"), "schema").validator_build = "legacy") - }, { name: "missing registration status", mutate: (value) => void delete objectField(objectField(value, "data"), "schema").registered diff --git a/packages/artifacts/test/schema.test.ts b/packages/artifacts/test/schema.test.ts index a6694a9b3..e67e59a92 100644 --- a/packages/artifacts/test/schema.test.ts +++ b/packages/artifacts/test/schema.test.ts @@ -178,7 +178,7 @@ test("current-controller schema materialization replaces only an older physical } }); -test("loads only a complete physical sealed schema bundle", () => { +test("loads a sealed schema bundle exactly as it was sealed, even when the installed build has moved on", () => { const root = mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ultrafuzz-sealed-schema-bundle-")); const destination = path.join(root, "schemas"); try { @@ -192,9 +192,32 @@ test("loads only a complete physical sealed schema bundle", () => { fs.symlinkSync(destination, linked, "dir"); assert.throws(() => artifactSchemaRegistryFromDirectory(linked), /snapshot directory is unsafe/u); + // #921: a run sealed by another build may hold a schema this build dropped, lack one it added, + // and carry an older `$id` for a file this build re-versioned. The bundle still loads as sealed; + // the artifact gate tells bundles apart by digest, not by agreement with the installed registry. fs.chmodSync(destination, 0o700); + const retired = { $schema: "https://json-schema.org/draft/2020-12/schema", $id: "urn:ultrafuzz:schema:retired:1" }; + fs.writeFileSync(path.join(destination, "retired.schema.json"), JSON.stringify(retired), "utf8"); + fs.chmodSync(path.join(destination, "report.schema.json"), 0o600); + fs.rmSync(path.join(destination, "report.schema.json")); + const reversioned = path.join(destination, "usage-ledger.schema.json"); + fs.chmodSync(reversioned, 0o600); + const usageLedger = JSON.parse(fs.readFileSync(reversioned, "utf8")) as Record; + fs.writeFileSync(reversioned, JSON.stringify({ ...usageLedger, $id: `${String(usageLedger.$id)}-sealed` }), "utf8"); + const sealed = artifactSchemaRegistryFromDirectory(destination); + assert.equal(sealed.find((entry) => entry.filename === "retired.schema.json")?.id, retired.$id); + assert.equal( + sealed.find((entry) => entry.filename === "usage-ledger.schema.json")?.id, + `${String(usageLedger.$id)}-sealed` + ); + assert.equal( + sealed.some((entry) => entry.filename === "report.schema.json"), + false + ); + assert.notEqual(schemaRegistryBundleDigest(sealed), artifactSchemaBundleDigest()); + fs.writeFileSync(path.join(destination, "foreign.schema.json"), "{}\n", "utf8"); - assert.throws(() => artifactSchemaRegistryFromDirectory(destination), /registry mismatch/u); + assert.throws(() => artifactSchemaRegistryFromDirectory(destination), /must declare Draft 2020-12/u); } finally { fs.chmodSync(destination, 0o700); for (const file of readdirSync(destination)) fs.chmodSync(path.join(destination, file), 0o600); @@ -2171,27 +2194,32 @@ test("planned graph v4 validates whole documents and executes every registered d ...artifactContractSchemaBinding("ultrafuzz/findings@2"), primary: true }; - const preUpgradeValidatorBuild = - "ultrafuzz-json-validator.v1:028be3251e9ac213ad6e1c037d8da47c9d903e8be563c8f4fa2149839565bcab"; - assert.equal(VALIDATOR_BUILD_IDENTITY, preUpgradeValidatorBuild); - const historicalBundle = structuredClone(graph); - historicalBundle.nodes = [ + // #921: a graph planned by another build records that build's contract digest, schema content and + // validator build. None of it is re-derived against the reading build, so a rebuild or upgrade + // after launch cannot make the run's own graph unreadable. + const otherBuild = structuredClone(graph); + otherBuild.nodes = [ { ...node, outputs: [ { ...findingsOutput, + contract_digest: "d".repeat(64), + schema_sha256: "e".repeat(64), schema_bundle_sha256: "f".repeat(64), - validator_build: preUpgradeValidatorBuild + validator_build: `ultrafuzz-json-validator.v1:${"9".repeat(64)}` } ] } ]; - assert.throws(() => assertPlannedGraph(historicalBundle), /schema binding changed/u); - assert.deepEqual(assertSealedPlannedGraph(historicalBundle), historicalBundle); - const historicalSchemaDrift = structuredClone(historicalBundle); - historicalSchemaDrift.nodes[0]!.outputs[0]!.schema_sha256 = "e".repeat(64); - assert.throws(() => assertSealedPlannedGraph(historicalSchemaDrift), /schema binding changed/u); + assert.notEqual(otherBuild.nodes[0]?.outputs[0]?.validator_build, VALIDATOR_BUILD_IDENTITY); + assert.deepEqual(assertPlannedGraph(otherBuild), otherBuild); + assert.deepEqual(assertSealedPlannedGraph(otherBuild), otherBuild); + // The shape still requires a schema binding exactly for schema-backed contracts. + const { schema_sha256: _schemaSha256, ...unboundFindings } = findingsOutput; + assert.equal(validatePlannedGraph({ ...graph, nodes: [{ ...node, outputs: [unboundFindings] }] }).ok, false); + const boundMarkdown = { ...node.outputs[0], ...artifactContractSchemaBinding("ultrafuzz/findings@2") }; + assert.equal(validatePlannedGraph({ ...graph, nodes: [{ ...node, outputs: [boundMarkdown] }] }).ok, false); const documentGateFailures: Array<{ name: string; graph: PlannedGraphDocument; message: RegExp }> = [ { @@ -2276,14 +2304,6 @@ test("planned graph v4 validates whole documents and executes every registered d graph: { ...graph, nodes: [{ ...node, loop: { ...node.loop, index: 1, count: 1, attempt_index: 1 } }] }, message: /inconsistent loop coordinates/u }, - { - name: "planned-graph-contract-identity", - graph: { - ...graph, - nodes: [{ ...node, outputs: [{ ...findingsOutput, contract_digest: "f".repeat(64) }] }] - }, - message: /contract digest changed/u - }, { name: "planned-graph-model-loop-coupling", graph: { diff --git a/packages/artifacts/test/semantic-gates.test.ts b/packages/artifacts/test/semantic-gates.test.ts index dbdded71b..1967b7fa4 100644 --- a/packages/artifacts/test/semantic-gates.test.ts +++ b/packages/artifacts/test/semantic-gates.test.ts @@ -11,7 +11,6 @@ import { MAX_SEMANTIC_GATE_ISSUES, SEMANTIC_GATE_REGISTRY, SEMANTIC_GATE_SCOPES, - artifactContractDefinition, executeOfflineSchemaSemanticGates, executeSchemaSemanticGates, executeSemanticGate, @@ -45,79 +44,6 @@ function reportWithCompletion(completion: unknown, runId = "run-a"): unknown { return { run_metadata: { run_id: runId }, completion }; } -const validPlannedOutput = { - path: "report.md", - contract: "ultrafuzz/nonempty-markdown@1", - contract_digest: artifactContractDefinition("ultrafuzz/nonempty-markdown@1").digest, - primary: true -}; - -const validPlannedNode = { - id: "node-a", - artifact_dir: "artifacts/node-a", - depends_on: [] as string[], - outputs: [validPlannedOutput], - loop: { index: 0, count: 1, attempt_index: 0 }, - model_fanout: [] as unknown[] -}; - -const validPlannedGraph = { nodes: [validPlannedNode] }; - -const smithersManifestOutput = { - path: validPlannedOutput.path, - contract: validPlannedOutput.contract, - contractDigest: validPlannedOutput.contract_digest, - primary: validPlannedOutput.primary -}; - -const smithersIdentityTask = { - attemptId: "attempt-a", - concreteNodeId: "node-a", - logicalNodeId: "logical-a", - smithersNodeId: "node:attempt-a", - verifierSmithersNodeId: "verify:attempt-a", - agentRef: "agent-a", - agentChain: [{ profileId: "profile-a", agentRef: "agent-a", role: "primary" }], - dependencies: [] as string[], - dependencySmithersNodeIds: [] as string[], - timeoutMs: 1_000, - heartbeatTimeoutMs: 500, - retries: 0, - artifactDir: "artifacts/node-a", - execution: { mode: "local", resources: { cpu: 1, memoryMiB: 512, timeoutSeconds: 1 } }, - metadata: { - run: { ultrafuzzRunId: "run-a", smithersWorkflowName: "workflow-a" }, - node: { - attemptId: "attempt-a", - concreteNodeId: "node-a", - logicalNodeId: "logical-a", - label: "Node A" - }, - model: { - agentRef: "agent-a", - agentChain: [{ profileId: "profile-a", agentRef: "agent-a", role: "primary" }] - }, - dependencies: { attemptIds: [] as string[], smithersNodeIds: [] as string[], concreteNodeIds: [] as string[] }, - timeout: { milliseconds: 1_000, seconds: 1, heartbeatTimeoutMs: 500 }, - retryPolicy: { maxAttempts: 1, sameAgentAttempts: 1, smithersRetries: 0 }, - execution: { mode: "local", resources: { cpu: 1, memoryMiB: 512, timeoutSeconds: 1 } }, - artifacts: { dir: "artifacts/node-a", outputs: [smithersManifestOutput] }, - loop: { index: 0, count: 1, mode: "parallel", attemptIndex: 0 } - } -}; - -const pinnedSubmoduleExpectation = { - schema_version: "ultrafuzz.pinned-submodules-expectation.v1", - source_commit: "a".repeat(40), - source_tree: "b".repeat(40), - manifest_sha256: "c".repeat(64), - top_level_roots: ["vendor/dependency"], - recursive_gitlinks: [{ path: "vendor/dependency", commit: "d".repeat(40), tree: "e".repeat(40) }], - entry_count: 1, - file_count: 0, - total_file_bytes: 0 -}; - const analysisBundleAccountingGatePositive = { run_count: 1, accounted_run_count: 1, @@ -450,18 +376,6 @@ const fixtures = { positive: { surfaces: [{ surface_id: "a" }], coverage_notes: [{ surface_id: "a" }] }, negative: { surfaces: [{ surface_id: "a" }], coverage_notes: [{ surface_id: "missing" }] } }, - "agent-source-proof-ref-uniqueness": { - positive: { refs: [{ name: "refs/heads/a" }] }, - negative: { refs: [{ name: "refs/heads/a" }, { name: "refs/heads/a" }] } - }, - "agent-source-proof-dependency-lineage": { - positive: { commit: "a".repeat(40), tree: "b".repeat(40), dependencies: pinnedSubmoduleExpectation }, - negative: { - commit: "a".repeat(40), - tree: "b".repeat(40), - dependencies: { ...pinnedSubmoduleExpectation, source_commit: "f".repeat(40) } - } - }, "aggregation-count-coupling": { positive: { source_generated_tests: 2, @@ -689,34 +603,6 @@ const fixtures = { finished_at: "2026-01-01T00:00:01Z" } }, - "artifact-manifest-file-path-uniqueness": { - positive: { files: [{ path: "a" }] }, - negative: { files: [{ path: "a" }, { path: "a" }] } - }, - "artifact-manifest-output-path-uniqueness": { - positive: { output_contracts: [{ path: "a" }] }, - negative: { output_contracts: [{ path: "a" }, { path: "a" }] } - }, - "artifact-manifest-prerequisite-node-uniqueness": { - positive: { prerequisite_manifests: [{ node_id: "a" }] }, - negative: { prerequisite_manifests: [{ node_id: "a" }, { node_id: "a" }] } - }, - "artifact-verification-artifact-path-uniqueness": { - positive: { artifacts: [{ path: "a" }] }, - negative: { artifacts: [{ path: "a" }, { path: "a" }] } - }, - "artifact-verification-exactly-one-primary": { - positive: { artifacts: [{ primary: true }] }, - negative: { artifacts: [{ primary: false }] } - }, - "artifact-verification-publication-digest-correspondence": { - positive: { artifacts: [{ path: "a", sha256: "1" }], publications: [{ path: "a", sha256: "1" }] }, - negative: { artifacts: [{ path: "a", sha256: "1" }], publications: [{ path: "a", sha256: "2" }] } - }, - "artifact-verification-publication-path-uniqueness": { - positive: { publications: [{ path: "a" }] }, - negative: { publications: [{ path: "a" }, { path: "a" }] } - }, "attempt-order": { positive: { lifecycle: { started_at: "2026-01-01T00:00:00Z", finished_at: "2026-01-01T00:00:01Z" } }, negative: { lifecycle: { started_at: "2026-01-01T00:00:01Z", finished_at: "2026-01-01T00:00:00Z" } } @@ -741,20 +627,6 @@ const fixtures = { positive: { backend_results: [{ fuzzer_backend: "a" }] }, negative: { backend_results: [{ fuzzer_backend: "a" }, { fuzzer_backend: "a" }] } }, - "config-redactions-path-key-equality": { - positive: { - entries: [{ path: ["models", "profiles", "default", "model"], key: "models.profiles.default.model" }] - }, - negative: { entries: [{ path: ["models", "profiles", "default", "model"], key: "wrong.path" }] } - }, - "config-redactions-path-uniqueness": { - positive: { - entries: [{ path: ["models", "profiles", "a", "model"] }, { path: ["models", "profiles", "b", "model"] }] - }, - negative: { - entries: [{ path: ["models", "profiles", "a", "model"] }, { path: ["models", "profiles", "a", "model"] }] - } - }, "coverage-evidence-reconciliation": { positive: validCoverageEvidence, negative: { @@ -932,47 +804,6 @@ const fixtures = { positive: { records: [{ dedupe_key: "a" }] }, negative: { records: [{ dedupe_key: "a" }, { dedupe_key: "a" }] } }, - "finding-campaign-provenance-coherence": { - positive: { - property_ids: ["property-1"], - fuzzer_backend: "recon", - contributing_backend_failures: [ - { - fuzzer_backend: "recon", - failure_id: "failure-1", - raw_result_ref: "recon-fuzzer-results.json" - } - ], - deduplication: { pre_dedup_count: 1 } - }, - negative: { - property_ids: ["property-1"], - fuzzer_backend: "medusa", - contributing_backend_failures: [ - { - fuzzer_backend: "recon", - failure_id: "failure-1", - raw_result_ref: "recon-fuzzer-results.json" - } - ], - deduplication: { pre_dedup_count: 1 } - } - }, - "finding-evidence-span-consistency": { - positive: { - evidence: [{ line: 4, end_line: 8 }, { line_ranges: [{ line: 10, end_line: 12 }, { line: 14 }] }] - }, - negative: { evidence: [{ line: 8, end_line: 4 }] } - }, - "finding-projected-reference-uniqueness": { - positive: { family_variants: [{ id: "a", dedupe_key: "a" }] }, - negative: { - family_variants: [ - { id: "a", dedupe_key: "a" }, - { id: "a", dedupe_key: "b" } - ] - } - }, "findings-id-uniqueness": { positive: [{ id: "a" }], negative: [{ id: "a" }, { id: "a" }] @@ -1074,97 +905,6 @@ const fixtures = { positive: { files: [{ path: "a" }] }, negative: { files: [{ path: "a" }, { path: "a" }] } }, - "invariant-suite-file-path-uniqueness": { - positive: { files: [{ path: "a" }] }, - negative: { files: [{ path: "a" }, { path: "a" }] } - }, - "invariant-suite-file-tombstone-disjointness": { - positive: { files: [{ path: "a" }], tombstones: ["b"] }, - negative: { files: [{ path: "a" }], tombstones: ["a"] } - }, - "invariant-suite-tombstone-uniqueness": { - positive: { tombstones: ["a"] }, - negative: { tombstones: ["a", "a"] } - }, - "planned-graph-acyclicity": { - positive: validPlannedGraph, - negative: { - nodes: [ - { ...validPlannedNode, id: "a", depends_on: ["b"] }, - { ...validPlannedNode, id: "b", depends_on: ["a"] } - ] - } - }, - "planned-graph-artifact-dir-identity": { - positive: validPlannedGraph, - negative: { nodes: [{ ...validPlannedNode, artifact_dir: "artifacts/other" }] } - }, - "planned-graph-contract-identity": { - positive: validPlannedGraph, - negative: { - nodes: [{ ...validPlannedNode, outputs: [{ ...validPlannedOutput, contract_digest: "0".repeat(64) }] }] - } - }, - "planned-graph-dependency-join": { - positive: validPlannedGraph, - negative: { nodes: [{ ...validPlannedNode, depends_on: ["missing"] }] } - }, - "planned-graph-exactly-one-primary": { - positive: validPlannedGraph, - negative: { nodes: [{ ...validPlannedNode, outputs: [{ ...validPlannedOutput, primary: false }] }] } - }, - "planned-graph-loop-coupling": { - positive: validPlannedGraph, - negative: { nodes: [{ ...validPlannedNode, loop: { index: 1, count: 1, attempt_index: 1 } }] } - }, - "planned-graph-model-fanout-uniqueness": { - positive: validPlannedGraph, - negative: { - nodes: [ - { - ...validPlannedNode, - model_fanout: [ - { model_profile_id: "m", model_index: 0, loop_index: 0, attempt_index: 0 }, - { model_profile_id: "m", model_index: 0, loop_index: 0, attempt_index: 0 } - ] - } - ] - } - }, - "planned-graph-model-loop-coupling": { - positive: validPlannedGraph, - negative: { - nodes: [ - { - ...validPlannedNode, - model_fanout: [{ model_profile_id: "m", model_index: 0, loop_index: 1, attempt_index: 0 }] - } - ] - } - }, - "planned-graph-node-id-uniqueness": { - positive: validPlannedGraph, - negative: { nodes: [validPlannedNode, { ...validPlannedNode }] } - }, - "planned-graph-output-path-uniqueness": { - positive: validPlannedGraph, - negative: { - nodes: [{ ...validPlannedNode, outputs: [validPlannedOutput, { ...validPlannedOutput, primary: false }] }] - } - }, - "planned-graph-workflow-node-join": { - positive: { nodes: [{ ...validPlannedNode, workflow: { node_id: "a", task_node_ids: ["a"] } }] }, - negative: { nodes: [{ ...validPlannedNode, workflow: { node_id: "a", task_node_ids: ["b"] } }] } - }, - "planned-graph-workflow-task-uniqueness": { - positive: { nodes: [{ ...validPlannedNode, workflow: { node_id: "a", task_node_ids: ["a"] } }] }, - negative: { - nodes: [ - { ...validPlannedNode, workflow: { node_id: "a", task_node_ids: ["a"] } }, - { ...validPlannedNode, id: "node-b", workflow: { node_id: "a", task_node_ids: ["a"] } } - ] - } - }, "property-campaign-coverage-metric-uniqueness": { positive: { coverage: { metrics: [{ name: "branches" }] } }, negative: { coverage: { metrics: [{ name: "branches" }, { name: "branches" }] } } @@ -1308,28 +1048,6 @@ const fixtures = { } } }, - "run-metadata-accounting-workflow-identity": { - positive: { - workflow: { run_id: "workflow-a" }, - accounting: { workflow_run_id: "workflow-a", current: { workflow_run_id: "workflow-a" } } - }, - negative: { - workflow: { run_id: "workflow-a" }, - accounting: { workflow_run_id: "workflow-b", current: { workflow_run_id: "workflow-b" } } - } - }, - "run-metadata-current-segment-equality": { - positive: { accounting: { current: { total_tokens: 2 }, segments: [{ total_tokens: 1 }, { total_tokens: 2 }] } }, - negative: { accounting: { current: { total_tokens: 1 }, segments: [{ total_tokens: 1 }, { total_tokens: 2 }] } } - }, - "run-metadata-workflow-id-equality": { - positive: { workflow_ids: ["workflow-a"], workflow: { run_id: "workflow-a" } }, - negative: { workflow_ids: ["workflow-b"], workflow: { run_id: "workflow-a" } } - }, - "run-plan-attempt-id-uniqueness": { - positive: { rendered_prompts: [{ attempt_id: "a" }, { attempt_id: "b" }] }, - negative: { rendered_prompts: [{ attempt_id: "a" }, { attempt_id: "a" }] } - }, "run-state-node-key-equality": { positive: { nodes: { a: { node_id: "a" } } }, negative: { nodes: { a: { node_id: "b" } } } @@ -1360,75 +1078,6 @@ const fixtures = { positive: [{ severity: "Medium", impact: "High", likelihood: "Low" }], negative: [{ severity: "High", impact: "High", likelihood: "Low" }] }, - "smithers-task-attempt-id-uniqueness": { - positive: { tasks: [{ attemptId: "a" }] }, - negative: { tasks: [{ attemptId: "a" }, { attemptId: "a" }] } - }, - "smithers-task-workflow-id-uniqueness": { - positive: { tasks: [{ smithersNodeId: "node:a", verifierSmithersNodeId: "verify:a" }] }, - negative: { - tasks: [ - { smithersNodeId: "node:a", verifierSmithersNodeId: "verify:a" }, - { smithersNodeId: "node:a", verifierSmithersNodeId: "verify:b" } - ] - } - }, - "smithers-task-document-identity": { - positive: { run_id: "run-a", workflow_name: "workflow-a", tasks: [smithersIdentityTask] }, - negative: { - run_id: "run-a", - workflow_name: "workflow-a", - tasks: [{ ...smithersIdentityTask, smithersNodeId: "node:wrong" }] - } - }, - "smithers-task-pinned-submodule-expectation": { - positive: { pinned_submodules: pinnedSubmoduleExpectation, tasks: [{ execution: { mode: "cloud" } }] }, - negative: { - pinned_submodules: { ...pinnedSubmoduleExpectation, file_count: 2 }, - tasks: [{ execution: { mode: "cloud" } }] - } - }, - "smithers-task-dependency-join": { - positive: { - tasks: [ - { attemptId: "a", verifierSmithersNodeId: "verify:a", dependencies: [], dependencySmithersNodeIds: [] }, - { - attemptId: "b", - verifierSmithersNodeId: "verify:b", - dependencies: ["a"], - dependencySmithersNodeIds: ["verify:a"] - } - ] - }, - negative: { - tasks: [ - { - attemptId: "b", - verifierSmithersNodeId: "verify:b", - dependencies: ["missing"], - dependencySmithersNodeIds: ["verify:missing"] - } - ] - } - }, - "smithers-task-dependency-acyclicity": { - positive: { - tasks: [ - { attemptId: "a", verifierSmithersNodeId: "verify:a", dependencySmithersNodeIds: [] }, - { attemptId: "b", verifierSmithersNodeId: "verify:b", dependencySmithersNodeIds: ["verify:a"] } - ] - }, - negative: { - tasks: [ - { attemptId: "a", verifierSmithersNodeId: "verify:a", dependencySmithersNodeIds: ["verify:b"] }, - { attemptId: "b", verifierSmithersNodeId: "verify:b", dependencySmithersNodeIds: ["verify:a"] } - ] - } - }, - "source-run-not-self": { - positive: { run_id: "run-new", source_run_id: "run-source" }, - negative: { run_id: "run-same", source_run_id: "run-same" } - }, "strategy-detection-dedupe-key-uniqueness": { positive: [{ dedupe_key: "a" }], negative: [{ dedupe_key: "a" }, { dedupe_key: "a" }] @@ -1563,12 +1212,6 @@ test("severity and report vocabulary gates reject renamed reachability tokens at }); test("property references alone do not claim fuzzer campaign provenance", () => { - assert.equal( - executeSemanticGate("finding-campaign-provenance-coherence", { - document: { property_ids: ["property-1"] } - }).status, - "passed" - ); assert.equal( executeSemanticGate("findings-campaign-provenance-coherence", { document: [{ property_ids: ["property-1"] }] @@ -1576,24 +1219,26 @@ test("property references alone do not claim fuzzer campaign provenance", () => "passed" ); assert.equal( - executeSemanticGate("finding-campaign-provenance-coherence", { - document: { property_ids: ["property-1"], fuzzer_backend: "recon" } + executeSemanticGate("findings-campaign-provenance-coherence", { + document: [{ property_ids: ["property-1"], fuzzer_backend: "recon" }] }).status, "passed" ); assert.equal( - executeSemanticGate("finding-campaign-provenance-coherence", { - document: { property_ids: ["property-1"], fuzzer_backend: "recon", deduplication: { pre_dedup_count: 1 } } + executeSemanticGate("findings-campaign-provenance-coherence", { + document: [{ property_ids: ["property-1"], fuzzer_backend: "recon", deduplication: { pre_dedup_count: 1 } }] }).status, "passed" ); assert.equal( - executeSemanticGate("finding-campaign-provenance-coherence", { - document: { - property_ids: ["property-1"], - fuzzer_backend: "recon", - contributing_backend_failures: [{ fuzzer_backend: "recon", failure_id: "failure-1" }] - } + executeSemanticGate("findings-campaign-provenance-coherence", { + document: [ + { + property_ids: ["property-1"], + fuzzer_backend: "recon", + contributing_backend_failures: [{ fuzzer_backend: "recon", failure_id: "failure-1" }] + } + ] }).status, "failed" ); @@ -1755,13 +1400,6 @@ test("workspace patch path gate reports exact nonduplicated field diagnostics", test("canonical finding span semantics run for every embedding schema", () => { const cases = [ - { - filename: "finding.schema.json", - gate: "finding-evidence-span-consistency", - positive: { evidence: [{ line: 3, end_line: 5 }] }, - negative: { evidence: [{ line: 5, end_line: 3 }] }, - path: "$.evidence[0].end_line" - }, { filename: "findings.schema.json", gate: "findings-evidence-span-consistency", @@ -2642,12 +2280,10 @@ test("every contextual registration executes real positive and negative checks", const root = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ultrafuzz-semantic-gates-")); try { fs.mkdirSync(path.join(root, "generated-tests")); - fs.writeFileSync(path.join(root, "artifact.json"), "artifact\n"); fs.writeFileSync(path.join(root, "generated-tests", "test.sol"), "test\n"); fs.writeFileSync(path.join(root, "generated-tests", "helper.sol"), "helper\n"); fs.writeFileSync(path.join(root, "generated-tests", "binary.dat"), Buffer.from([0xff])); fs.writeFileSync(path.join(root, "copied.sol"), "test\n"); - const digest = crypto.createHash("sha256").update("artifact\n").digest("hex"); const contentDigest = crypto.createHash("sha256").update("snapshot", "utf8").digest("hex"); const aggregationSourceBytes = Buffer.from("test\n", "utf8"); const aggregationSourceDigest = crypto.createHash("sha256").update(aggregationSourceBytes).digest("hex"); @@ -2685,7 +2321,7 @@ test("every contextual registration executes real positive and negative checks", const campaignEvidenceDigest = crypto.createHash("sha256").update(campaignEvidenceBytes).digest("hex"); const campaignTimeoutCommand = "timeout --preserve-status --signal=INT --kill-after=300s 60s recon fuzz . --workers 1 " + - "--timeout 60 --test-limit 18446744073709551615"; + "--timeout 60 --test-limit 18446744073709551615 --seq-len 100"; const campaignTimeoutPlan = { configured_fuzzer_timeout_seconds: 60, recon_internal_timeout_seconds: 60, @@ -2695,6 +2331,7 @@ test("every contextual registration executes real positive and negative checks", finalization_reserve_seconds: 100, configured_budget_seconds: 460, recon_test_limit: "18446744073709551615", + recon_sequence_length: 100, backend_started_at: "2026-01-01T00:00:00.000Z", fuzzing_deadline_utc: "2026-01-01T00:01:00.000Z", force_kill_deadline_utc: "2026-01-01T00:06:00.000Z", @@ -2723,16 +2360,6 @@ test("every contextual registration executes real positive and negative checks", Exclude, { positive: unknown; negative: unknown; context: SemanticGateContext } > = { - "agent-source-proof-commit-binding": { - positive: { commit: "c", tree: "t", refs: [{ name: "r", object: "o" }] }, - negative: { commit: "wrong", tree: "t", refs: [{ name: "r", object: "o" }] }, - context: { git: { commit: "c", tree: "t", refs: { r: "o" } } } - }, - "analysis-bundle-file-digest": { - positive: { files: [{ path: "artifact.json", sha256: digest, size_bytes: 9 }] }, - negative: { files: [{ path: "artifact.json", sha256: "0".repeat(64), size_bytes: 9 }] }, - context: { filesystem: { rootDirectory: root } } - }, "analysis-bundle-inclusion-omission-coverage": { positive: { omissions: [ @@ -2795,16 +2422,6 @@ test("every contextual registration executes real positive and negative checks", } } }, - "artifact-manifest-file-digest": { - positive: { files: [{ path: "artifact.json", sha256: digest, size_bytes: 9 }] }, - negative: { files: [{ path: "missing.json", sha256: digest, size_bytes: 9 }] }, - context: { filesystem: { rootDirectory: root } } - }, - "artifact-verification-plan-contract-identity": { - positive: { node_id: "a", artifacts: [{ ...validPlannedOutput }] }, - negative: { node_id: "b", artifacts: [{ ...validPlannedOutput }] }, - context: { plannedGraph: { node: { id: "a", outputs: [{ ...validPlannedOutput }] } } } - }, "attempt-reuse-source-link": { positive: { workflow_run_id: "workflow-current", @@ -3047,7 +2664,7 @@ test("every contextual registration executes real positive and negative checks", context: { artifactSet: { campaignPlan: campaignTimeoutPlan, - campaignSummary: { outcome: "complete" } + campaignSummary: { outcome: "complete", sequence_length: 100 } }, propertyCampaignTimeout: { configuredFuzzerTimeoutSeconds: 60, @@ -3317,7 +2934,6 @@ test("every contextual registration executes real positive and negative checks", schemaId: "schema-id", schemaSha256: "a".repeat(64), schemaBundleSha256: "b".repeat(64), - validatorBuild: "validator-build", artifactSha256: "c".repeat(64) } } @@ -3543,108 +3159,6 @@ test("every contextual registration executes real positive and negative checks", } } }, - "run-state-fingerprint": { - positive: { graph_fingerprint: "g", config_fingerprint: "c" }, - negative: { graph_fingerprint: "wrong", config_fingerprint: "c" }, - context: { runtimeState: { graphFingerprint: "g", configFingerprint: "c" } } - }, - "smithers-task-planned-graph-coverage": { - positive: { - tasks: [{ attemptId: "node-a", concreteNodeId: "node-a" }] - }, - negative: { tasks: [] }, - context: { - plannedGraph: { - document: { - nodes: [ - { - ...validPlannedNode, - logical_id: "logical-a", - display_name: "Node A", - kind: "agentic" - } - ] - } - } - } - }, - "smithers-task-planned-graph-identity": { - positive: { - tasks: [ - { - attemptId: "node-a", - concreteNodeId: "node-a", - logicalNodeId: "logical-a", - metadata: { - node: { logicalNodeId: "logical-a", label: "Node A" }, - loop: { index: 0, count: 1, mode: "parallel", attemptIndex: 0 }, - artifacts: { outputs: [smithersManifestOutput] } - } - } - ] - }, - negative: { - tasks: [ - { - attemptId: "node-a", - concreteNodeId: "node-a", - logicalNodeId: "wrong", - metadata: { - node: { logicalNodeId: "wrong", label: "Node A" }, - loop: { index: 0, count: 1, mode: "parallel", attemptIndex: 0 }, - artifacts: { outputs: [smithersManifestOutput] } - } - } - ] - }, - context: { - plannedGraph: { - document: { - nodes: [ - { - ...validPlannedNode, - logical_id: "logical-a", - display_name: "Node A", - kind: "agentic", - loop: { index: 0, count: 1, mode: "parallel", attempt_index: 0 } - } - ] - } - } - } - }, - "smithers-task-planned-graph-dependency-join": { - positive: { - tasks: [ - { - concreteNodeId: "node-a", - dependencies: ["node-b"], - dependencySmithersNodeIds: ["verify:node-b"], - metadata: { dependencies: { concreteNodeIds: ["node-b"] } } - } - ] - }, - negative: { - tasks: [ - { - concreteNodeId: "node-a", - dependencies: [], - dependencySmithersNodeIds: [], - metadata: { dependencies: { concreteNodeIds: [] } } - } - ] - }, - context: { - plannedGraph: { - document: { - nodes: [ - { ...validPlannedNode, id: "node-a", depends_on: ["node-b"], kind: "agentic" }, - { ...validPlannedNode, id: "node-b", depends_on: [], kind: "agentic" } - ] - } - } - } - }, "usage-ledger-event-order": { positive: { workflow_run_id: "workflow-a", source_event_sequence: 2, control_generation: "a".repeat(64) }, negative: { workflow_run_id: "workflow-a", source_event_sequence: 0, control_generation: "a".repeat(64) }, @@ -3734,170 +3248,6 @@ test("every contextual registration executes real positive and negative checks", } }); -test("Smithers planned dependency semantics omit unresolved dynamic groups only", () => { - const pendingGraph = { - nodes: [ - { - ...validPlannedNode, - id: "consumer", - depends_on: ["fanout"], - dynamic_dependencies: ["fanout"], - kind: "agentic" - }, - { - ...validPlannedNode, - id: "fanout", - depends_on: [], - kind: "agentic", - dynamic: { status: "pending" } - } - ] - }; - const pendingTask = { - tasks: [ - { - concreteNodeId: "consumer", - dependencies: [], - dependencySmithersNodeIds: [], - metadata: { dependencies: { concreteNodeIds: [] } } - } - ] - }; - - assert.equal( - executeSemanticGate("smithers-task-planned-graph-dependency-join", { - document: pendingTask, - context: { plannedGraph: { document: pendingGraph } } - }).status, - "passed" - ); - - const compiledPendingTask = { - tasks: [ - { - ...structuredClone(pendingTask.tasks[0]!), - metadata: { dependencies: { concreteNodeIds: ["fanout"] } } - } - ] - }; - assert.equal( - executeSemanticGate("smithers-task-planned-graph-dependency-join", { - document: compiledPendingTask, - context: { plannedGraph: { document: pendingGraph } } - }).status, - "passed" - ); - - const twoPendingGraph = { - nodes: [ - { - ...structuredClone(pendingGraph.nodes[0]!), - depends_on: ["fanout", "fanout-second"], - dynamic_dependencies: ["fanout", "fanout-second"] - }, - structuredClone(pendingGraph.nodes[1]!), - { - ...structuredClone(pendingGraph.nodes[1]!), - id: "fanout-second" - } - ] - }; - assert.equal( - executeSemanticGate("smithers-task-planned-graph-dependency-join", { - document: compiledPendingTask, - context: { plannedGraph: { document: twoPendingGraph } } - }).status, - "failed" - ); - - const materializedGraph = { - nodes: [ - { ...structuredClone(pendingGraph.nodes[0]!), depends_on: ["generated"] }, - { - ...structuredClone(pendingGraph.nodes[1]!), - dynamic: { status: "expanded" } - }, - { ...validPlannedNode, id: "generated", depends_on: [], kind: "agentic" } - ] - }; - const materializedTask = { - tasks: [ - { - concreteNodeId: "consumer", - dependencies: ["generated"], - dependencySmithersNodeIds: ["verify:generated"], - metadata: { dependencies: { concreteNodeIds: ["generated"] } } - } - ] - }; - assert.equal( - executeSemanticGate("smithers-task-planned-graph-dependency-join", { - document: materializedTask, - context: { plannedGraph: { document: materializedGraph } } - }).status, - "passed" - ); - - const staleExpandedGraph = structuredClone(materializedGraph); - staleExpandedGraph.nodes.find((node) => node.id === "consumer")!.depends_on = ["fanout"]; - assert.equal( - executeSemanticGate("smithers-task-planned-graph-dependency-join", { - document: pendingTask, - context: { plannedGraph: { document: staleExpandedGraph } } - }).status, - "failed" - ); - - const alignedStaleExpandedTask = { - tasks: [ - { - ...structuredClone(pendingTask.tasks[0]!), - dependencies: ["fanout"], - dependencySmithersNodeIds: ["verify:fanout"], - metadata: { dependencies: { concreteNodeIds: ["fanout"] } } - } - ] - }; - const alignedStaleExpanded = executeSemanticGate("smithers-task-planned-graph-dependency-join", { - document: alignedStaleExpandedTask, - context: { plannedGraph: { document: staleExpandedGraph } } - }); - assert.equal(alignedStaleExpanded.status, "failed"); - assert.ok( - alignedStaleExpanded.status === "failed" && - alignedStaleExpanded.issues.some((entry) => - /retains expanded dynamic dependency placeholder/u.test(entry.message) - ) - ); - - const ordinaryGraph = { - nodes: [ - { ...structuredClone(pendingGraph.nodes[0]!), depends_on: ["fanout", "ordinary"] }, - structuredClone(pendingGraph.nodes[1]!), - { ...validPlannedNode, id: "ordinary", depends_on: [], kind: "agentic" } - ] - }; - assert.equal( - executeSemanticGate("smithers-task-planned-graph-dependency-join", { - document: pendingTask, - context: { plannedGraph: { document: ordinaryGraph } } - }).status, - "failed" - ); - - const missingGraph = { - nodes: [{ ...structuredClone(pendingGraph.nodes[0]!), depends_on: ["missing"] }] - }; - const missing = executeSemanticGate("smithers-task-planned-graph-dependency-join", { - document: pendingTask, - context: { plannedGraph: { document: missingGraph } } - }); - assert.equal(missing.status, "failed"); - assert.ok( - missing.status === "failed" && missing.issues.some((entry) => /planned dependency node/u.test(entry.message)) - ); -}); - test("strict final reports preserve dropped false positives as exactly one non-production row", () => { const finding = { id: "finding-a", @@ -6445,30 +5795,43 @@ test("campaign evidence-file closure names duplicated, missing, and unreferenced ); }); -test("campaign timeout evidence reports the expected final artifact deadline next to a forged execution deadline", () => { - const command = - "timeout --preserve-status --signal=INT --kill-after=300s 60s recon fuzz . --workers 1 " + - "--timeout 60 --test-limit 18446744073709551615"; - const plan = { - configured_fuzzer_timeout_seconds: 60, - recon_internal_timeout_seconds: 60, - host_soft_timeout_seconds: 60, - host_force_kill_grace_seconds: 300, - artifact_finalization_reserve_seconds: 100, - finalization_reserve_seconds: 100, - configured_budget_seconds: 460, - recon_test_limit: "18446744073709551615", - backend_started_at: "2026-01-01T00:00:00.000Z", - fuzzing_deadline_utc: "2026-01-01T00:01:00.000Z", - force_kill_deadline_utc: "2026-01-01T00:06:00.000Z", - final_artifact_deadline_utc: "2026-01-01T00:07:40.000Z", - deadline: "2026-01-01T00:07:40.000Z", - backend: { exact_shell_escaped_command: command }, - command_plan: [{ phase: "campaign", command }] - }; - const result = executeSemanticGate("property-campaign-timeout-evidence", { +const CAMPAIGN_TIMEOUT_EVIDENCE_COMMAND = + "timeout --preserve-status --signal=INT --kill-after=300s 60s recon fuzz . --workers 1 " + + "--timeout 60 --test-limit 18446744073709551615 --seq-len 100"; + +interface CampaignTimeoutEvidenceFields { + plan: Record; + document: Record; + summary: Record; +} + +/** property-campaign-timeout-evidence issues for a consistent 60-second Recon campaign after `mutate`. */ +function campaignTimeoutEvidenceIssues( + mutate: (fields: CampaignTimeoutEvidenceFields) => void = () => undefined +): { path: string; message: string }[] { + const command = CAMPAIGN_TIMEOUT_EVIDENCE_COMMAND; + const fields: CampaignTimeoutEvidenceFields = { + plan: { + configured_fuzzer_timeout_seconds: 60, + recon_internal_timeout_seconds: 60, + host_soft_timeout_seconds: 60, + host_force_kill_grace_seconds: 300, + artifact_finalization_reserve_seconds: 100, + finalization_reserve_seconds: 100, + configured_budget_seconds: 460, + recon_test_limit: "18446744073709551615", + recon_sequence_length: 100, + backend_started_at: "2026-01-01T00:00:00.000Z", + fuzzing_deadline_utc: "2026-01-01T00:01:00.000Z", + force_kill_deadline_utc: "2026-01-01T00:06:00.000Z", + final_artifact_deadline_utc: "2026-01-01T00:07:40.000Z", + deadline: "2026-01-01T00:07:40.000Z", + backend: { exact_shell_escaped_command: command }, + command_plan: [{ phase: "campaign", command }] + }, document: { configured_timeout_seconds: 60, + sequence_length: 100, exact_command: command, start_timestamp: "2026-01-01T00:00:00.000Z", end_timestamp: "2026-01-01T00:01:00.000Z", @@ -6480,13 +5843,16 @@ test("campaign timeout evidence reports the expected final artifact deadline nex usable_results: true, started_at: "2026-01-01T00:00:00.000Z", finished_at: "2026-01-01T00:01:00.000Z", - // The incident's wrong-field bug: copied from the plan's - // fuzzing_deadline_utc instead of final_artifact_deadline_utc. - deadline: "2026-01-01T00:01:00.000Z" + deadline: "2026-01-01T00:07:40.000Z" } }, + summary: { outcome: "complete", sequence_length: 100 } + }; + mutate(fields); + const result = executeSemanticGate("property-campaign-timeout-evidence", { + document: fields.document, context: { - artifactSet: { campaignPlan: plan, campaignSummary: { outcome: "complete" } }, + artifactSet: { campaignPlan: fields.plan, campaignSummary: fields.summary }, propertyCampaignTimeout: { configuredFuzzerTimeoutSeconds: 60, plannedTimeoutSeconds: 600, @@ -6494,11 +5860,17 @@ test("campaign timeout evidence reports the expected final artifact deadline nex } } }); + assert.notEqual(result.status, "requires-context"); + return result.status === "failed" ? result.issues.map((entry) => ({ path: entry.path, message: entry.message })) : []; +} - assert.equal(result.status, "failed"); - const issues = result.status === "failed" ? result.issues : []; +test("campaign timeout evidence reports the expected final artifact deadline next to a forged execution deadline", () => { assert.deepEqual( - issues.map((entry) => ({ path: entry.path, message: entry.message })), + campaignTimeoutEvidenceIssues(({ document }) => { + // The incident's wrong-field bug: copied from the plan's + // fuzzing_deadline_utc instead of final_artifact_deadline_utc. + (document.execution as Record).deadline = "2026-01-01T00:01:00.000Z"; + }), [ { path: "$.execution.deadline", @@ -6510,6 +5882,56 @@ test("campaign timeout evidence reports the expected final artifact deadline nex ); }); +test("campaign timeout evidence requires the stateful Recon sequence length wherever it is recorded", () => { + const issuePaths = (mutate?: (fields: CampaignTimeoutEvidenceFields) => void) => + campaignTimeoutEvidenceIssues(mutate).map((entry) => entry.path); + const withCommand = ({ plan, document }: CampaignTimeoutEvidenceFields, next: string) => { + plan.backend = { exact_shell_escaped_command: next }; + plan.command_plan = [{ phase: "campaign", command: next }]; + document.exact_command = next; + (document.execution as Record).command = next; + }; + + assert.deepEqual(issuePaths(), []); + // The result record's sequence_length is optional in its schema. + assert.deepEqual( + issuePaths(({ document }) => { + delete document.sequence_length; + }), + [] + ); + assert.deepEqual( + issuePaths(({ plan }) => { + plan.recon_sequence_length = 1; + }), + ["$.campaign_plan_ref#recon_sequence_length"] + ); + assert.deepEqual( + issuePaths(({ summary }) => { + summary.sequence_length = 1; + }), + ["$.campaign_summary_ref#sequence_length"] + ); + assert.deepEqual( + issuePaths(({ document }) => { + document.sequence_length = 1; + }), + ["$.sequence_length"] + ); + const command = CAMPAIGN_TIMEOUT_EVIDENCE_COMMAND; + for (const next of [ + command.replace("--seq-len 100", "--seq-len 1"), + command.replace(" --seq-len 100", ""), + `${command} --seq-len 100` + ]) { + assert.deepEqual( + issuePaths((fields) => withCommand(fields, next)), + ["$.exact_command"], + next + ); + } +}); + test("campaign context joins report the expected bare declared path next to a node-dir reference", () => { const positive = { campaign_plan_ref: "campaign-plan.json", diff --git a/packages/artifacts/test/smithers-task-manifest.test.ts b/packages/artifacts/test/smithers-task-manifest.test.ts index 4590a2679..1bcf1e6cc 100644 --- a/packages/artifacts/test/smithers-task-manifest.test.ts +++ b/packages/artifacts/test/smithers-task-manifest.test.ts @@ -428,7 +428,7 @@ test("rejects missing graph coverage, extra tasks, and graph dependency drift", ); }); -test("only review tasks treat inputs from continuing groups as optional", () => { +test("optional task inputs must come from producers that continue on failure", () => { const continuingGraph = graph(); continuingGraph.groups = { specialists: { defaults: { failure_policy: "continue" } }, @@ -482,21 +482,27 @@ test("only review tasks treat inputs from continuing groups as optional", () => withDependencies(dependent("reviewer", reviewerOptional), "review") ]); - // A continuing specialist's own inputs stay required (#1132's stateful topology). - assert.doesNotThrow(() => - assertSmithersTaskManifestMatchesPlannedGraph(current([], ["/runs/run-1/artifacts/producer"]), continuingGraph) - ); - assert.throws( - () => - assertSmithersTaskManifestMatchesPlannedGraph( - current(["/runs/run-1/artifacts/producer"], ["/runs/run-1/artifacts/producer"]), - continuingGraph - ), - /"consumer" optional dependency artifact directories/u - ); + const producer = "/runs/run-1/artifacts/producer"; + + // Which consumers opt in is compiler policy, so the gate accepts every shape it has emitted: only + // the review task opts in (#1120), every consumer opts in (before #1120), or a review task keeps the + // input required (dynamic lowering of an empty expansion keeps its source required). + const emittedShapes: Array<[consumerOptional: string[], reviewerOptional: string[]]> = [ + [[], [producer]], + [[producer], [producer]], + [[], []] + ]; + for (const [consumerOptional, reviewerOptional] of emittedShapes) { + assert.doesNotThrow(() => + assertSmithersTaskManifestMatchesPlannedGraph(current(consumerOptional, reviewerOptional), continuingGraph) + ); + } + // A producer that halts on failure can never become optional. + const haltingGraph = structuredClone(continuingGraph); + haltingGraph.groups.specialists = {}; assert.throws( - () => assertSmithersTaskManifestMatchesPlannedGraph(current([], []), continuingGraph), - /"reviewer" optional dependency artifact directories/u + () => assertSmithersTaskManifestMatchesPlannedGraph(current([], [producer]), haltingGraph), + /task "reviewer" marks dependency "producer" optional, but that producer does not continue on failure/u ); }); diff --git a/packages/artifacts/test/threat-goal-artifacts.test.ts b/packages/artifacts/test/threat-goal-artifacts.test.ts index 1c0df73cb..0611e65b8 100644 --- a/packages/artifacts/test/threat-goal-artifacts.test.ts +++ b/packages/artifacts/test/threat-goal-artifacts.test.ts @@ -9,11 +9,9 @@ import { GOAL_PLAN_POLICY, GOAL_PLAN_SCHEMA_VERSION, THREAT_MODEL_SCHEMA_VERSION, - buildFindingSourceExpectations, goalPlanExpansionFacts, materializeCanonicalThreatModelMarkdown, renderThreatModelMarkdown, - normalizeFindings, validateArtifactContract, validateGoalPlan, validateThreatModel, @@ -724,383 +722,6 @@ test("an applicable class remains planned when no explicit threat maps to it", ( assert.equal(result.value?.class_goals[0]?.coverage_gap, true); }); -test("initial finding normalization ignores agent provenance and seeds the runtime-controlled producer", () => { - const artifactDir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-goal-provenance-")); - fs.writeFileSync( - path.join(artifactDir, "findings.json"), - JSON.stringify([ - { - title: "Overdue liquidation bypass", - status: "candidate", - severity_guess: "high", - confidence: "high", - summary: "The position can be liquidated before its required overdue boundary.", - source_node_id: "legacy-node", - source_nodes: ["dynamic:class:liquidation:fixed-term-before-overdue", "legacy-node"] - } - ]) - ); - - const result = normalizeFindings({ - artifactDir, - nodeId: "filesystem-safe-attempt", - provenance: { producerNodeId: "dynamic:threat:liquidation:overdue" } - }); - - assert.equal(result.findings[0]?.source_node_id, "dynamic:threat:liquidation:overdue"); - assert.equal(result.findings[0]?.producer_node_id, "dynamic:threat:liquidation:overdue"); - assert.deepEqual(result.findings[0]?.source_nodes, ["dynamic:threat:liquidation:overdue"]); -}); - -test("downstream normalization preserves discovery sources without treating the transformer as a discoverer", () => { - const artifactDir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-goal-provenance-transform-")); - fs.writeFileSync( - path.join(artifactDir, "deduped-findings.json"), - JSON.stringify([ - { - title: "Overdue liquidation bypass", - status: "candidate", - severity_guess: "high", - confidence: "high", - summary: "Two focused hunters corroborated the same root cause.", - source_nodes: ["dynamic:threat:liquidation:overdue", "dynamic:class:liquidation:fixed-term-before-overdue"] - } - ]) - ); - - const result = normalizeFindings({ - artifactDir, - relativePath: "deduped-findings.json", - provenance: { producerNodeId: "dedupe-findings" }, - preserveSourceNodes: true, - requireSourceNodes: true, - allowedSourceNodes: ["dynamic:threat:liquidation:overdue", "dynamic:class:liquidation:fixed-term-before-overdue"] - }); - assert.equal(result.findings[0]?.producer_node_id, "dedupe-findings"); - assert.equal(result.findings[0]?.source_node_id, "dynamic:threat:liquidation:overdue"); - assert.deepEqual(result.findings[0]?.source_nodes, [ - "dynamic:threat:liquidation:overdue", - "dynamic:class:liquidation:fixed-term-before-overdue" - ]); - - const tamperedDir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-goal-provenance-tamper-")); - fs.writeFileSync( - path.join(tamperedDir, "deduped-findings.json"), - JSON.stringify([ - { - title: "Invented source", - status: "candidate", - severity_guess: "high", - confidence: "high", - summary: "The review node tried to invent discovery provenance.", - source_nodes: ["dynamic:threat:liquidation:invented"] - } - ]) - ); - assert.throws( - () => - normalizeFindings({ - artifactDir: tamperedDir, - relativePath: "deduped-findings.json", - provenance: { producerNodeId: "dedupe-findings" }, - preserveSourceNodes: true, - requireSourceNodes: true, - allowedSourceNodes: ["dynamic:threat:liquidation:overdue"] - }), - /not present in dependency findings/u - ); -}); - -test("dedupe provenance rejects dropped corroborating sources and incomplete lifecycle coverage", () => { - const upstream = [ - { - node_id: "dynamic:threat:liquidation:overdue", - artifact_path: "artifacts/threat/findings.json", - finding: { - id: "threat-finding", - dedupe_key: "raw:threat", - producer_node_id: "dynamic:threat:liquidation:overdue", - source_nodes: ["dynamic:threat:liquidation:overdue"] - } - }, - { - node_id: "dynamic:class:liquidation:fixed-term-before-overdue", - artifact_path: "artifacts/class/findings.json", - finding: { - id: "class-finding", - dedupe_key: "raw:class", - producer_node_id: "dynamic:class:liquidation:fixed-term-before-overdue", - source_nodes: ["dynamic:class:liquidation:fixed-term-before-overdue"] - } - } - ]; - const lifecycleLedger = { - schema_version: "1.0", - records: [ - { - dedupe_key: "root:fixed-term-overdue", - source_artifacts: [ - { - node_id: "dynamic:threat:liquidation:overdue", - finding_id: "threat-finding" - }, - { - node_id: "dynamic:class:liquidation:fixed-term-before-overdue", - finding_id: "class-finding" - } - ] - } - ] - }; - const expectations = buildFindingSourceExpectations({ - upstream, - lifecycleLedger, - requireLifecycleCoverage: true - }); - const artifactDir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-goal-provenance-drop-")); - fs.writeFileSync( - path.join(artifactDir, "deduped-findings.json"), - JSON.stringify([ - { - id: "kept-finding", - dedupe_key: "root:fixed-term-overdue", - title: "Fixed-term liquidation before overdue", - status: "candidate", - severity_guess: "high", - confidence: "high", - summary: "Two focused hunters found the same lifecycle boundary bug.", - source_nodes: ["dynamic:threat:liquidation:overdue"] - } - ]) - ); - assert.throws( - () => - normalizeFindings({ - artifactDir, - relativePath: "deduped-findings.json", - provenance: { producerNodeId: "dedupe-findings" }, - preserveSourceNodes: true, - requireSourceNodes: true, - allowedSourceNodes: upstream.flatMap((entry) => entry.finding.source_nodes), - sourceExpectations: expectations, - requireSourceExpectation: true - }), - /does not preserve the exact dependency discovery-source union/u - ); - - const complete = JSON.parse(fs.readFileSync(path.join(artifactDir, "deduped-findings.json"), "utf8")) as Array< - Record - >; - complete[0]!.source_nodes = [ - "dynamic:class:liquidation:fixed-term-before-overdue", - "dynamic:threat:liquidation:overdue" - ]; - fs.writeFileSync(path.join(artifactDir, "deduped-findings.json"), JSON.stringify(complete)); - const normalized = normalizeFindings({ - artifactDir, - relativePath: "deduped-findings.json", - provenance: { producerNodeId: "dedupe-findings" }, - preserveSourceNodes: true, - requireSourceNodes: true, - allowedSourceNodes: upstream.flatMap((entry) => entry.finding.source_nodes), - sourceExpectations: expectations, - requireSourceExpectation: true - }); - assert.deepEqual(normalized.findings[0]?.source_nodes, [ - "dynamic:threat:liquidation:overdue", - "dynamic:class:liquidation:fixed-term-before-overdue" - ]); - - const incompleteLedger = structuredClone(lifecycleLedger); - incompleteLedger.records[0]!.source_artifacts.pop(); - assert.throws( - () => - buildFindingSourceExpectations({ - upstream, - lifecycleLedger: incompleteLedger, - requireLifecycleCoverage: true - }), - /omitted dependency findings/u - ); -}); - -test("generic finding ID collisions cannot union unrelated discovery lanes", () => { - const upstream = [ - { - node_id: "dynamic:threat:accounting:rounding", - artifact_path: "artifacts/threat/findings.json", - finding: { - id: "finding-1", - upstream_id: "threat-original", - dedupe_key: "raw:threat", - source_nodes: ["dynamic:threat:accounting:rounding"] - } - }, - { - node_id: "dynamic:class:authorization:roles", - artifact_path: "artifacts/class/findings.json", - finding: { - id: "finding-1", - upstream_id: "class-original", - dedupe_key: "raw:class", - source_nodes: ["dynamic:class:authorization:roles"] - } - } - ]; - const lifecycleLedger = { - schema_version: "1.0", - records: [ - { - dedupe_key: "root:rounding", - family_id: "family:rounding", - source_artifacts: [ - { - node_id: "dynamic:threat:accounting:rounding", - finding_id: "finding-1" - } - ] - }, - { - dedupe_key: "root:roles", - family_id: "family:roles", - source_artifacts: [ - { - node_id: "dynamic:class:authorization:roles", - finding_id: "finding-1" - } - ] - } - ] - }; - const expectations = buildFindingSourceExpectations({ - upstream, - lifecycleLedger, - requireLifecycleCoverage: true - }); - const artifactDir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-provenance-id-collision-")); - const finding = { - id: "finding-1", - dedupe_key: "root:rounding", - title: "Rounding drift", - status: "candidate", - severity_guess: "high", - confidence: "high", - summary: "The rounding lane found a loss of accounting precision.", - source_nodes: ["dynamic:threat:accounting:rounding"] - }; - fs.writeFileSync(path.join(artifactDir, "deduped-findings.json"), JSON.stringify([finding])); - const normalized = normalizeFindings({ - artifactDir, - relativePath: "deduped-findings.json", - provenance: { producerNodeId: "dedupe-findings" }, - preserveSourceNodes: true, - requireSourceNodes: true, - allowedSourceNodes: upstream.flatMap((entry) => entry.finding.source_nodes), - sourceExpectations: expectations, - requireSourceExpectation: true - }); - assert.deepEqual(normalized.findings[0]?.source_nodes, ["dynamic:threat:accounting:rounding"]); - - for (const [label, contradictoryIdentity] of [ - ["family ID", { family_id: "family:roles" }], - ["finding reference", { upstream_id: "class-original" }] - ] as const) { - fs.writeFileSync( - path.join(artifactDir, "deduped-findings.json"), - JSON.stringify([{ ...finding, ...contradictoryIdentity }]) - ); - assert.throws( - () => - normalizeFindings({ - artifactDir, - relativePath: "deduped-findings.json", - provenance: { producerNodeId: "dedupe-findings" }, - preserveSourceNodes: true, - requireSourceNodes: true, - allowedSourceNodes: upstream.flatMap((entry) => entry.finding.source_nodes), - sourceExpectations: expectations, - requireSourceExpectation: true - }), - new RegExp(`${label} conflicts with higher-priority dependency provenance`, "u") - ); - } - - const freshOutputIdentity = { - ...finding, - id: "kept-finding-new", - family_id: "family:new", - source_nodes: ["dynamic:threat:accounting:rounding"] - }; - fs.writeFileSync(path.join(artifactDir, "deduped-findings.json"), JSON.stringify([freshOutputIdentity])); - assert.deepEqual( - normalizeFindings({ - artifactDir, - relativePath: "deduped-findings.json", - provenance: { producerNodeId: "dedupe-findings" }, - preserveSourceNodes: true, - requireSourceNodes: true, - allowedSourceNodes: upstream.flatMap((entry) => entry.finding.source_nodes), - sourceExpectations: expectations, - requireSourceExpectation: true - }).findings[0]?.source_nodes, - ["dynamic:threat:accounting:rounding"] - ); - - const ambiguous = { - ...finding, - dedupe_key: undefined, - source_nodes: upstream.flatMap((entry) => entry.finding.source_nodes) - }; - fs.writeFileSync(path.join(artifactDir, "deduped-findings.json"), JSON.stringify([ambiguous])); - assert.throws( - () => - normalizeFindings({ - artifactDir, - relativePath: "deduped-findings.json", - provenance: { producerNodeId: "dedupe-findings" }, - preserveSourceNodes: true, - requireSourceNodes: true, - allowedSourceNodes: upstream.flatMap((entry) => entry.finding.source_nodes), - sourceExpectations: expectations, - requireSourceExpectation: true - }), - /finding ID matches conflicting dependency provenance records/u - ); - - const forgedStrongIdentity = { - ...finding, - dedupe_key: "root:unknown", - source_nodes: upstream.flatMap((entry) => entry.finding.source_nodes) - }; - fs.writeFileSync(path.join(artifactDir, "deduped-findings.json"), JSON.stringify([forgedStrongIdentity])); - assert.throws( - () => - normalizeFindings({ - artifactDir, - relativePath: "deduped-findings.json", - provenance: { producerNodeId: "dedupe-findings" }, - preserveSourceNodes: true, - requireSourceNodes: true, - allowedSourceNodes: upstream.flatMap((entry) => entry.finding.source_nodes), - sourceExpectations: expectations, - requireSourceExpectation: true - }), - /dedupe key does not match any dependency provenance record/u - ); - - const duplicateKeyLedger = structuredClone(lifecycleLedger); - duplicateKeyLedger.records[1]!.dedupe_key = "root:rounding"; - assert.throws( - () => - buildFindingSourceExpectations({ - upstream, - lifecycleLedger: duplicateKeyLedger, - requireLifecycleCoverage: true - }), - /duplicate dedupe_key/u - ); -}); - test("selected vulnerability-class snapshots are manifest-backed exact artifacts", () => { const artifactDir = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-selected-classes-")); const contents = Buffer.from("# Share inflation\n\nFocused hunter instructions.\n"); @@ -1348,6 +969,34 @@ test("goal-plan replacements are bounded titles, not inlined JSON records", () = assert.equal(validateGoalPlan(withReplacement(true)).ok, true); }); +test("goal-plan replacement values cannot carry template placeholders into the dynamic render", () => { + const plan = goalPlanFixture(); + const threatId = "liquidation:overdue"; + const withReplacement = (value: string): Record => { + const next = structuredClone(plan); + const [goal] = next.threat_goals as Array>; + assert.ok(goal); + goal.replacements = { [threatId]: value }; + return next; + }; + + // The renderer resolves `{{...}}` inside a replacement value, and an unbound name such as `amount` + // throws inside the workflow render. Item references and escaped braces are template syntax too, + // which a label never needs, so the verifier rejects every form rather than re-deriving the renderer. + for (const label of [ + "Late tick lets {{amount}} round down", + "Overdue liquidation of {{item.id}}", + "Late tick lets \\{{amount}} round down" + ]) { + const rejected = validateGoalPlan(withReplacement(label)); + assert.equal(rejected.ok, false, label); + assert.match(rejected.issues.map((issue) => issue.message).join("; "), /must not contain template braces/u); + } + + // Single braces are prose, as in the bounded-titles test above. + assert.equal(validateGoalPlan(withReplacement("{withdraw} settles before the late tick")).ok, true); +}); + test("escaped required goal placeholders are rejected instead of rendering as literals", () => { const plan = goalPlanFixture(); const goal = (plan.threat_goals as Array>)[0]!; @@ -1412,56 +1061,3 @@ function classGoalPlanFixture(selectedPath: string): Record { sealGoalPlanCardinality(plan); return plan; } - -test("a cross-node dedupe root only resolves through its lifecycle ledger record", () => { - // Regression for the workflow-sync path, which rebuilt these expectations - // without the ledger. Without it the union expectation does not exist at all, - // so a legitimate merge matches nothing and a succeeded node is reported as a - // task-output-validation-failure. - const upstream = [ - { - node_id: "dynamic:threat:liquidation:overdue", - artifact_path: "/runs/r/artifacts/a/findings.json", - finding: { id: "f-threat", source_nodes: ["dynamic:threat:liquidation:overdue"] } - }, - { - node_id: "dynamic:class:liquidation:fixed-term-before-overdue", - artifact_path: "/runs/r/artifacts/b/findings.json", - finding: { id: "f-class", source_nodes: ["dynamic:class:liquidation:fixed-term-before-overdue"] } - } - ]; - const lifecycleLedger = { - schema_version: "1.0", - records: [ - { - dedupe_key: "root:fixed-term-overdue", - source_artifacts: [ - { path: "a/findings.json", node_id: "dynamic:threat:liquidation:overdue", finding_id: "f-threat" }, - { - path: "b/findings.json", - node_id: "dynamic:class:liquidation:fixed-term-before-overdue", - finding_id: "f-class" - } - ] - } - ] - }; - const unionOf = (expectations: ReturnType): string[][] => - expectations.filter((expectation) => expectation.source_nodes.length > 1).map((e) => [...e.source_nodes].sort()); - - assert.deepEqual(unionOf(buildFindingSourceExpectations({ upstream, requireLifecycleCoverage: true })), []); - assert.deepEqual( - unionOf(buildFindingSourceExpectations({ upstream, lifecycleLedger, requireLifecycleCoverage: true })), - [["dynamic:class:liquidation:fixed-term-before-overdue", "dynamic:threat:liquidation:overdue"]] - ); - - // The ledger keys the union expectation, so the retained finding must carry - // that same key. Both prompts now say so explicitly. - const withLedger = buildFindingSourceExpectations({ upstream, lifecycleLedger, requireLifecycleCoverage: true }); - assert.ok( - withLedger.some( - (expectation) => - expectation.finding_keys.includes("root:fixed-term-overdue") && expectation.source_nodes.length === 2 - ) - ); -}); diff --git a/packages/cli/package.json b/packages/cli/package.json index 5b3a2fd9e..22684ed68 100644 --- a/packages/cli/package.json +++ b/packages/cli/package.json @@ -26,7 +26,7 @@ "scripts": { "build": "rm -rf dist && tsc -p tsconfig.json && cp -R schema dist/schema && node scripts/verify-schema-registry.mjs", "test": "pnpm --filter @ultrafuzz/cli... build && rm -rf dist-test && tsc -p tsconfig.test.json && node --test --test-concurrency=1 dist-test/test/*.test.js", - "test:pr-smoke:prebuilt": "rm -rf dist-test && tsc -p tsconfig.test.json && node scripts/run-pr-smoke-tests.mjs", + "test:e2e": "pnpm --filter @ultrafuzz/cli... build && rm -rf dist-test && tsc -p tsconfig.test.json && node --test dist-test/test/e2e/*.test.js", "typecheck": "pnpm --filter @ultrafuzz/cli^... build && tsc -p tsconfig.json --noEmit --pretty false" }, "dependencies": { diff --git a/packages/cli/scripts/run-pr-smoke-tests.mjs b/packages/cli/scripts/run-pr-smoke-tests.mjs deleted file mode 100644 index 71fbe9af1..000000000 --- a/packages/cli/scripts/run-pr-smoke-tests.mjs +++ /dev/null @@ -1,62 +0,0 @@ -import { spawnSync } from "node:child_process"; -import { readFileSync } from "node:fs"; - -const namedTests = new Map([ - ["test/cli.test.ts", ["status surfaces a terminal product and live workflow lifecycle divergence"]], - [ - "test/cli-contracts.test.ts", - [ - "known command failures can retain a valid typed data snapshot", - "status separates successful queries, partial reports, unavailable reports, and unknown completion" - ] - ], - [ - "test/status-report.test.ts", - [ - "watch stops at ended runs with unavailable reports and distinguishes attention stops", - "status shows complete execution with a partial unchecked report without changing the execution verdict", - "status shows report failure and unknown coverage without presenting a false report path" - ] - ] -]); -const selectedTestNames = [...namedTests.values()].flat(); - -for (const [sourcePath, names] of namedTests) { - const source = readFileSync(sourcePath, "utf8"); - for (const name of names) { - if (!source.includes(`test(${JSON.stringify(name)}`)) { - throw new Error(`PR CLI smoke test is not registered: ${name}`); - } - } -} - -const pattern = `^(?:${selectedTestNames.map(escapeRegExp).join("|")})$`; -const files = [...namedTests.keys()].map((sourcePath) => `dist-test/${sourcePath.replace(/\.ts$/u, ".js")}`); -const result = spawnSync( - process.execPath, - ["--test", "--test-concurrency=1", "--test-reporter=tap", `--test-name-pattern=${pattern}`, ...files], - { - encoding: "utf8", - stdio: ["inherit", "pipe", "pipe"] - } -); -process.stdout.write(result.stdout ?? ""); -process.stderr.write(result.stderr ?? ""); -if (result.error !== undefined) throw result.error; -if (result.status !== 0) process.exit(result.status ?? 1); -assertNamedTestsPassed(result.stdout ?? "", selectedTestNames); - -function escapeRegExp(value) { - return value.replace(/[.*+?^${}()|[\]\\]/gu, "\\$&"); -} - -function assertNamedTestsPassed(output, expectedTestNames) { - const passedTestNames = [...output.matchAll(/^ok [0-9]+ - (.+)$/gmu)] - .map((match) => match[1]) - .filter((name) => !name.includes(" # SKIP") && !name.includes(" # TODO")) - .sort(); - const expected = [...expectedTestNames].sort(); - if (passedTestNames.length !== expected.length || passedTestNames.some((name, index) => name !== expected[index])) { - throw new Error(`PR CLI smoke passed unexpected tests: ${JSON.stringify(passedTestNames)}`); - } -} diff --git a/packages/cli/src/commands/doctor.ts b/packages/cli/src/commands/doctor.ts index 21192c00a..d0eb9315c 100644 --- a/packages/cli/src/commands/doctor.ts +++ b/packages/cli/src/commands/doctor.ts @@ -34,7 +34,7 @@ function renderDoctor(value: DoctorValue): string { `- ${entry.name}: ${ entry.available ? `${entry.path ?? "available"}${entry.version == null ? "" : ` (${entry.version})`}` - : "missing from execution environment" + : `missing from execution environment${entry.required ? "" : " (not required)"}` }` ), "Workflow engine:", diff --git a/packages/cli/src/commands/eval/history.ts b/packages/cli/src/commands/eval/history.ts index 24724f243..a8d441b54 100644 --- a/packages/cli/src/commands/eval/history.ts +++ b/packages/cli/src/commands/eval/history.ts @@ -2,7 +2,6 @@ import path from "node:path"; import { Args, Command, Flags } from "@oclif/core"; import { - assertEvalHistoryRecency, checkEvalHistoryCharts, publishEvalRunToHistory, readEvalHistory, @@ -36,11 +35,7 @@ export default class EvalHistory extends Command { "benchmark-policy-root": Flags.string({ summary: "Candidate checkout whose benchmark manifests define the published run" }), - check: Flags.boolean({ summary: "Validate history and fail when checked-in charts are stale" }), - "max-age-days": Flags.integer({ - summary: "Fail when the newest observation is older than this many days", - min: 1 - }) + check: Flags.boolean({ summary: "Validate history and fail when checked-in charts are stale" }) }; async run(): Promise { @@ -51,7 +46,6 @@ export default class EvalHistory extends Command { try { if (args.evalRunId !== undefined) { if (flags.check) throw new Error("--check cannot append an eval run"); - if (flags["max-age-days"] !== undefined) throw new Error("--max-age-days cannot append an eval run"); if ( flags.benchmark === undefined || flags.lane === undefined || @@ -96,9 +90,6 @@ export default class EvalHistory extends Command { } const history = readEvalHistory(historyPath); - if (flags["max-age-days"] !== undefined) { - assertEvalHistoryRecency({ history, maxAgeDays: flags["max-age-days"], now: new Date() }); - } if (flags.check) { const mismatches = checkEvalHistoryCharts(history, chartsDirectory); if (mismatches.length > 0) throw new Error(`eval history charts are stale: ${mismatches.join(", ")}`); diff --git a/packages/cli/src/commands/eval/run.ts b/packages/cli/src/commands/eval/run.ts index 0b63aa0e4..9fc09bb39 100644 --- a/packages/cli/src/commands/eval/run.ts +++ b/packages/cli/src/commands/eval/run.ts @@ -13,7 +13,7 @@ import { import { toCliEvalRunData } from "../../cli-contracts.js"; export default class EvalRun extends Command { - static override summary = "Launch Ultrafuzz runs for an eval suite matrix and stream node telemetry"; + static override summary = "Launch Ultrafuzz runs for an eval suite matrix and watch them to a terminal state"; static override flags = { ...globalFlags, suite: Flags.string({ summary: "Eval suite YAML path (defaults to [eval].eval_config)" }), @@ -26,7 +26,7 @@ export default class EvalRun extends Command { summary: "Maximum time to watch each launched row before returning", min: 1 }), - "no-watch": Flags.boolean({ summary: "Launch detached without polling runs or streaming node telemetry" }) + "no-watch": Flags.boolean({ summary: "Launch detached without polling runs" }) }; async run(): Promise { diff --git a/packages/cli/src/commands/report/bundle.ts b/packages/cli/src/commands/report/bundle.ts index a76fdf005..e0cff4858 100644 --- a/packages/cli/src/commands/report/bundle.ts +++ b/packages/cli/src/commands/report/bundle.ts @@ -5,18 +5,12 @@ import path from "node:path"; import { Args, Command, Flags } from "@oclif/core"; import { DEFAULT_STRICT_JSONL_MAX_BYTES, - DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES, - DEFAULT_STRICT_JSONL_MAX_RECORDS, - assertEventRecord, assertNoSymlinkComponents, assertPathInside, assertRegularFileInside, layoutForRunRoot, - parseStrictJsonBytes, + parseEventJournalBytes, readRegularFileSnapshot, - validateStrictJsonlHistory, - type EventRecord, - type StrictJsonlCodec, validateSafeId } from "@ultrafuzz/artifacts"; import { @@ -385,7 +379,7 @@ function loadValidatedEventJournalSnapshot( assertRegularFileInside(runRoot, eventsPath, "event journal"); const contents = readRegularFileSnapshot(eventsPath, DEFAULT_STRICT_JSONL_MAX_BYTES); - validateEventJournalSnapshot(contents, expectedRunId); + parseEventJournalBytes(contents, expectedRunId); return { absolutePath: eventsPath, archivePath: "events.jsonl", @@ -393,83 +387,6 @@ function loadValidatedEventJournalSnapshot( }; } -function validateEventJournalSnapshot(contents: Buffer, expectedRunId: string): void { - if (contents.byteLength === 0) return; - if (contents[contents.byteLength - 1] !== 0x0a) { - throw new Error("event journal has a torn or unterminated final record"); - } - - let text: string; - try { - text = new TextDecoder("utf-8", { fatal: true }).decode(contents); - } catch (error) { - throw new Error("event journal is not valid UTF-8", { cause: error }); - } - const lines = text.split("\n"); - lines.pop(); - if (lines.length > DEFAULT_STRICT_JSONL_MAX_RECORDS) { - throw new Error(`event journal exceeds the ${DEFAULT_STRICT_JSONL_MAX_RECORDS}-record limit`); - } - - const codec = eventJournalCodec(expectedRunId); - const records: EventRecord[] = []; - for (const [index, line] of lines.entries()) { - const lineNumber = index + 1; - if (line.trim().length === 0) throw new Error(`event journal contains a blank record at line ${lineNumber}`); - const lineBytes = Buffer.from(line, "utf8"); - if (lineBytes.byteLength > DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES) { - throw new Error( - `event journal record ${lineNumber} exceeds the ${DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES}-byte limit` - ); - } - let parsed: unknown; - try { - parsed = parseStrictJsonBytes(lineBytes, { - maxBytes: DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES, - maxDepth: 128, - maxItems: 100_000, - maxProperties: 100_000 - }); - } catch (error) { - throw new Error( - `event journal record ${lineNumber} is invalid strict JSON: ${error instanceof Error ? error.message : String(error)}`, - { cause: error } - ); - } - records.push(codec.parseRecord(parsed, `$[${index}]`)); - } - validateStrictJsonlHistory(records, codec); -} - -function eventJournalCodec(expectedRunId: string): StrictJsonlCodec { - return { - label: "event journal", - parseRecord: (value, recordPath) => { - const record = assertEventRecord(value, recordPath); - if (record.run_id !== expectedRunId) { - throw new Error( - `${recordPath}.run_id belongs to ${JSON.stringify(record.run_id)}, expected ${JSON.stringify(expectedRunId)}` - ); - } - return record; - }, - identity: (record) => record.event_id, - validateHistory: (records) => { - const firstRunId = records[0]?.run_id; - let priorTimestamp = records[0]?.timestamp; - for (const [index, record] of records.entries()) { - if (firstRunId !== undefined && record.run_id !== firstRunId) { - throw new Error(`event journal changes run_id at record ${index + 1}`); - } - if (priorTimestamp !== undefined && record.timestamp < priorTimestamp) { - throw new Error(`event journal timestamps are not ordered at record ${index + 1}`); - } - priorTimestamp = record.timestamp; - } - } - }; -} - function collectBundleFiles( runRoot: string, diagnostics: RuntimeDiagnostic[], diff --git a/packages/cli/src/commands/resume.ts b/packages/cli/src/commands/resume.ts index 2f3bb79da..44e701df8 100644 --- a/packages/cli/src/commands/resume.ts +++ b/packages/cli/src/commands/resume.ts @@ -5,6 +5,7 @@ import { cliEntrypoint, cliIo, commandFromRuntime, + diagnosticsText, emitCommandResult, globalFlags, projectRoot @@ -37,11 +38,14 @@ export default class Resume extends Command { resetNode: flags["reset-node"], env: cliIo().env }); - emitCommandResult( - this, - "resume", - commandFromRuntime("resume", result, (value) => `Submitted ${value.action}: ${value.workflow_run_id}\n`), - flags.json === true + const commandResult = commandFromRuntime("resume", result, (value) => + value.submitted + ? `Submitted ${value.action}: ${value.workflow_run_id}\n` + : `Run already active: ${value.workflow_run_id}; no new controller was started. If a pause is still draining, resume again once status reports paused; if its controller process just exited, resume again after 30 seconds.\n` ); + if (result.ok && result.diagnostics.length > 0) { + commandResult.text = `${commandResult.text ?? ""}${diagnosticsText(result.diagnostics)}`; + } + emitCommandResult(this, "resume", commandResult, flags.json); } } diff --git a/packages/cli/src/commands/stats.ts b/packages/cli/src/commands/stats.ts index fcbbe3f8c..a5ed23685 100644 --- a/packages/cli/src/commands/stats.ts +++ b/packages/cli/src/commands/stats.ts @@ -17,6 +17,7 @@ import { validateSafeId } from "@ultrafuzz/artifacts"; import { + TRANSIENT_SYNC_DIAGNOSTIC_CODES, describeObservationSynchronizationDeadline, observationSynchronizationDeadline, runsRootForProject, @@ -162,17 +163,9 @@ async function loadLocalEvidence( const snapshot = readCoherentLocalEvidenceSnapshot(layout); const runMetadata = assertRunMetadataDocument(parseLocalJson(snapshot.runMetadata, layout.runMetadataPath), runId); if (!synchronized.ok && runMetadata.workflow !== undefined) { - const transientCodes = new Set([ - "WORKFLOW_INSPECT_FAILED", - "WORKFLOW_INSPECT_INVALID", - "WORKFLOW_EVENTS_FAILED", - "WORKFLOW_EVENTS_INVALID", - "WORKFLOW_TOKEN_EVENTS_FAILED", - "WORKFLOW_TOKEN_EVENTS_INVALID", - "WORKFLOW_SYNC_CANCELLED", - "WORKFLOW_SYNC_DEADLINE_EXCEEDED" - ]); - const authorityFailure = synchronized.diagnostics.find((diagnostic) => !transientCodes.has(diagnostic.code)); + const authorityFailure = synchronized.diagnostics.find( + (diagnostic) => !TRANSIENT_SYNC_DIAGNOSTIC_CODES.has(diagnostic.code) + ); if (authorityFailure !== undefined) { throw new Error(`linked workflow authority is invalid: ${authorityFailure.message}`); } diff --git a/packages/cli/test/audit-profile-commands.test.ts b/packages/cli/test/audit-profile-commands.test.ts index ab3979741..b1022d489 100644 --- a/packages/cli/test/audit-profile-commands.test.ts +++ b/packages/cli/test/audit-profile-commands.test.ts @@ -1,12 +1,12 @@ import assert from "node:assert/strict"; import fs from "node:fs"; -import os from "node:os"; import path from "node:path"; -import test from "node:test"; +import test, { type TestContext } from "node:test"; import { packagedTopology } from "@ultrafuzz/config"; import { runCli } from "../src/index.js"; +import { temporaryRoot } from "./temporary-root.js"; interface Capture { stdout: string; @@ -14,8 +14,8 @@ interface Capture { code: number; } -function tempProject(): string { - return fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-cli-profile-")); +function tempProject(t: TestContext): string { + return temporaryRoot("ufz-cli-profile-", t); } async function cli(project: string, argv: string[]): Promise { @@ -44,8 +44,8 @@ function data(capture: Capture): Record { return (JSON.parse(capture.stdout) as { data: Record }).data; } -test("profile list and detail expose the catalog and effective project policy", async () => { - const project = tempProject(); +test("profile list and detail expose the catalog and effective project policy", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); assert.deepEqual( fs.readFileSync(path.join(project, ".ultrafuzz", "topology.yml")), @@ -84,8 +84,8 @@ test("profile list and detail expose the catalog and effective project policy", assert.equal(detailData.effective_settings.strategy_loops, 1); }); -test("topology list, show, and copy use the packaged assets safely", async () => { - const project = tempProject(); +test("topology list, show, and copy use the packaged assets safely", async (t) => { + const project = tempProject(t); const listed = await cli(project, ["topology", "list", "--json"]); assert.equal(listed.code, 0, listed.stderr); const listData = data(listed) as { @@ -128,8 +128,8 @@ test("topology list, show, and copy use the packaged assets safely", async () => assert.equal(fs.existsSync(path.join(project, "..", "outside.yml")), false); }); -test("validate accepts CLI profile and topology overrides with the documented precedence", async () => { - const project = tempProject(); +test("validate accepts CLI profile and topology overrides with the documented precedence", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); fs.writeFileSync(path.join(project, ".ultrafuzz", "topology.yml"), "not: [valid\n", "utf8"); diff --git a/packages/cli/test/cli.test.ts b/packages/cli/test/cli.test.ts index 18676ca3a..3b68dc9bf 100644 --- a/packages/cli/test/cli.test.ts +++ b/packages/cli/test/cli.test.ts @@ -4,7 +4,7 @@ import crypto from "node:crypto"; import fs from "node:fs"; import os from "node:os"; import path from "node:path"; -import test from "node:test"; +import test, { type TestContext } from "node:test"; import { ARTIFACT_VERIFICATION_SCHEMA_VERSION, @@ -31,7 +31,6 @@ import { type RunLayout } from "@ultrafuzz/artifacts"; import { DASHBOARD_HTTP_SCHEMA_VERSION, serveDashboard } from "@ultrafuzz/dashboard"; -import { NodeTelemetryPump, type EvalArtifactUpload, type EvalMatrixRow, type EvalReporter } from "@ultrafuzz/evals"; import { loadGoalSearchCoverageSnapshot, loadVerifiedRunOutputSnapshots, @@ -45,6 +44,7 @@ import AdmZip from "adm-zip"; import { validateReportBundleManifest } from "../src/cli-schema-registry.js"; import { runCli } from "../src/index.js"; import { formatStatusDuration } from "../src/status-rendering.js"; +import { temporaryRoot } from "./temporary-root.js"; interface Capture { stdout: string; @@ -52,8 +52,8 @@ interface Capture { code: number; } -function tempProject(): string { - return fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-cli-")); +function tempProject(t: TestContext): string { + return temporaryRoot("ufz-cli-", t); } function shellQuote(value: string): string { @@ -759,8 +759,8 @@ function writeCanonicalReportPair(reportDir: string, report: Record { - const project = tempProject(); +test("report render gives producers the exact canonical Markdown without host repair", async (t) => { + const project = tempProject(t); const reportPath = path.join(project, "report.json"); const markdownPath = path.join(project, "report.md"); const report = currentReport("producer-render", [currentReportIssue()]); @@ -778,8 +778,8 @@ test("report render gives producers the exact canonical Markdown without host re assert.deepEqual(fs.readFileSync(markdownPath), before); }); -test("report render renders goal coverage from the run-root census only through its flag", async () => { - const project = tempProject(); +test("report render renders goal coverage from the run-root census only through its flag", async (t) => { + const project = tempProject(t); const runId = "census-render"; const runRoot = path.join(project, "runs", runId); fs.mkdirSync(runRoot, { recursive: true }); @@ -1239,8 +1239,8 @@ function writeJsonRecord(filePath: string, value: Record): void fs.writeFileSync(filePath, `${JSON.stringify(value, null, 2)}\n`, "utf8"); } -test("init and validate emit schema-versioned launch JSON", async () => { - const project = tempProject(); +test("init and validate emit schema-versioned launch JSON", async (t) => { + const project = tempProject(t); fs.writeFileSync(path.join(project, "ultrafuzz.toml"), "# owned\n", "utf8"); const init = await cli(project, ["init", "--json"]); @@ -1280,25 +1280,28 @@ test("init and validate emit schema-versioned launch JSON", async () => { assert.match(tampered.stdout + tampered.stderr, /absent\.database/u); }); -test("plain init surfaces a customized stale agent adapter diagnostic", async () => { - const project = tempProject(); +test("plain init restores a customized agent adapter without touching project config", async (t) => { + const project = tempProject(t); const initial = await cli(project, ["init", "--force"]); assert.equal(initial.code, 0, initial.stderr); const adapterPath = path.join(project, ".smithers", "agents", "codex.ts"); - const customAdapter = 'export const customConfigPath = "ultrafuzz.toml";\n'; - fs.writeFileSync(adapterPath, customAdapter, "utf8"); + const stockAdapter = fs.readFileSync(adapterPath, "utf8"); + fs.writeFileSync(adapterPath, 'export const customConfigPath = "ultrafuzz.toml";\n', "utf8"); + const configPath = path.join(project, "ultrafuzz.toml"); + const customConfig = `${fs.readFileSync(configPath, "utf8")}\n# operator customization\n`; + fs.writeFileSync(configPath, customConfig, "utf8"); const result = await cli(project, ["init"]); assert.equal(result.code, 0, result.stderr); assert.equal(result.stderr, ""); - assert.match(result.stdout, /warning: INIT_AGENT_ADAPTER_UPDATE_REQUIRED:/u); - assert.match(result.stdout, /ULTRAFUZZ_CONFIG_PATH/u); - assert.equal(fs.readFileSync(adapterPath, "utf8"), customAdapter); + assert.doesNotMatch(result.stdout, /warning:/u); + assert.equal(fs.readFileSync(adapterPath, "utf8"), stockAdapter); + assert.equal(fs.readFileSync(configPath, "utf8"), customConfig); }); -test("run exposes the trusted reference expectation catalog option", async () => { - const project = tempProject(); +test("run exposes the trusted reference expectation catalog option", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeSmallTopology(project); const run = await cli( @@ -1316,8 +1319,8 @@ test("run exposes the trusted reference expectation catalog option", async () => ); }); -test("run input flags are explicit, strict, and never reinterpret malformed inline JSON as a path", async () => { - const project = tempProject(); +test("run input flags are explicit, strict, and never reinterpret malformed inline JSON as a path", async (t) => { + const project = tempProject(t); fs.writeFileSync(path.join(project, "looks-like-a-path.json"), '{"loaded":true}\n', "utf8"); const malformedInline = await cli(project, ["run", "--input-json", "looks-like-a-path.json", "--json"]); @@ -1352,8 +1355,8 @@ test("run input flags are explicit, strict, and never reinterpret malformed inli assert.match(JSON.stringify(parseJson(linkedInput).diagnostics), /cannot open regular file/iu); }); -test("run reads a bounded immutable workflow-input file", async () => { - const project = tempProject(); +test("run reads a bounded immutable workflow-input file", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeSmallTopology(project); fs.writeFileSync(path.join(project, "operator-input.json"), '{"ticket":3}\n', "utf8"); @@ -1376,8 +1379,8 @@ test("run reads a bounded immutable workflow-input file", async () => { assert.equal(savedConfig.retry.sameAgentAttempts, 3); }); -test("run, ps, status, inspect, report, materialize, clean, and lifecycle commands expose product workflow evidence", async () => { - const project = tempProject(); +test("run, ps, status, inspect, report, materialize, clean, and lifecycle commands expose product workflow evidence", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeReportTopology(project); @@ -1779,8 +1782,8 @@ test("run, ps, status, inspect, report, materialize, clean, and lifecycle comman assert.equal(pauseData.submitted, true); }); -test("status --watch --json keeps a failing poll on one NDJSON line", async () => { - const project = tempProject(); +test("status --watch --json keeps a failing poll on one NDJSON line", async (t) => { + const project = tempProject(t); const env = fakeSmithersEnv(project); const init = await cli(project, ["init", "--json"], env); assert.equal(init.code, 0, init.stderr); @@ -1803,8 +1806,8 @@ test("status --watch --json keeps a failing poll on one NDJSON line", async () = assert.equal((body.diagnostics as Array<{ code: string }>)[0]?.code, "WORKFLOW_CONTROL_EVIDENCE_INVALID"); }); -test("ps text prefers linked workflow terminal status over a stale local running projection", async () => { - const project = tempProject(); +test("ps text prefers linked workflow terminal status over a stale local running projection", async (t) => { + const project = tempProject(t); const env = fakeSmithersEnv(project); assert.equal((await cli(project, ["init", "--json"], env)).code, 0); writeSmallTopology(project); @@ -1846,8 +1849,8 @@ test("ps text prefers linked workflow terminal status over a stale local running assert.equal(jsonData.runs[0]?.workflow_status, "failed"); }); -test("status observes an incomplete launch with a successful CLI envelope and unknown liveness", async () => { - const project = tempProject(); +test("status observes an incomplete launch with a successful CLI envelope and unknown liveness", async (t) => { + const project = tempProject(t); const env = fakeSmithersEnv(project); assert.equal((await cli(project, ["init", "--json"], env)).code, 0); writeSmallTopology(project); @@ -1877,8 +1880,8 @@ test("status observes an incomplete launch with a successful CLI envelope and un assert.doesNotMatch(text.stdout, /Progress: 0%|ETA: 0|running-healthy/u); }); -test("status surfaces a terminal product and live workflow lifecycle divergence", async () => { - const project = tempProject(); +test("status surfaces a terminal product and live workflow lifecycle divergence", async (t) => { + const project = tempProject(t); const env = fakeSmithersEnv(project); assert.equal((await cli(project, ["init", "--json"], env)).code, 0); writeSmallTopology(project); @@ -1918,8 +1921,28 @@ test("status surfaces a terminal product and live workflow lifecycle divergence" ); }); -test("status --watch stops immediately on a degraded verdict even while product state is nonterminal", async () => { - const project = tempProject(); +test("resume of an already-active run says no controller was started instead of claiming a submission", async (t) => { + const project = tempProject(t); + const env = fakeSmithersEnv(project); + assert.equal((await cli(project, ["init", "--json"], env)).code, 0); + writeSmallTopology(project); + const runId = "resume-already-active"; + const run = await cli(project, ["run", "--run-id", runId, "--json"], env); + assert.equal(run.code, 0, run.stderr); + + // The fake runner still reports the run as running, so resume only attaches. + const resumed = await cli(project, ["resume", runId], env); + + assert.equal(resumed.code, 0, `${resumed.stderr}\n${resumed.stdout}`); + assert.match( + resumed.stdout, + /^Run already active: ultrafuzz-resume-already-active; no new controller was started\./mu + ); + assert.doesNotMatch(resumed.stdout, /Submitted/u); +}); + +test("status --watch stops immediately on a degraded verdict even while product state is nonterminal", async (t) => { + const project = tempProject(t); const env = fakeSmithersEnv(project); assert.equal((await cli(project, ["init", "--json"], env)).code, 0); writeSmallTopology(project); @@ -1940,8 +1963,8 @@ test("status --watch stops immediately on a degraded verdict even while product assert.equal(body.data?.report?.status, "unknown"); }); -test("status surfaces quota parking with preserved attempts and the resume remediation", async () => { - const project = tempProject(); +test("status surfaces quota parking with preserved attempts and the resume remediation", async (t) => { + const project = tempProject(t); const env = fakeSmithersEnv(project); assert.equal((await cli(project, ["init", "--json"], env)).code, 0); writeSmallTopology(project); @@ -1997,8 +2020,8 @@ test("status surfaces quota parking with preserved attempts and the resume remed assert.doesNotMatch(healthy.stdout, /^Quota:/mu); }); -test("old commands and backend flags are rejected instead of aliased or shimmed", async () => { - const project = tempProject(); +test("old commands and backend flags are rejected instead of aliased or shimmed", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); for (const argv of [ @@ -2018,8 +2041,8 @@ test("old commands and backend flags are rejected instead of aliased or shimmed" } }); -test("references status is restored and reports offline cache state", async () => { - const project = tempProject(); +test("references status is restored and reports offline cache state", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); const previousXdgCacheHome = process.env.XDG_CACHE_HOME; process.env.XDG_CACHE_HOME = path.join(project, "empty-cache"); @@ -2054,8 +2077,8 @@ test("references status is restored and reports offline cache state", async () = } }); -test("runtime command failures emit a failing exit code with JSON", async () => { - const project = tempProject(); +test("runtime command failures emit a failing exit code with JSON", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeSmallTopology(project); @@ -2068,8 +2091,8 @@ test("runtime command failures emit a failing exit code with JSON", async () => assert.match(JSON.stringify(body.diagnostics), /CONFIG_MODEL_AGENT_INVALID/); }); -test("run rejects an OpenRouter override whose effective model is invalid", async () => { - const project = tempProject(); +test("run rejects an OpenRouter override whose effective model is invalid", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeSmallTopology(project); @@ -2093,8 +2116,8 @@ test("run rejects an OpenRouter override whose effective model is invalid", asyn assert.equal(fs.existsSync(path.join(project, ".ultrafuzz", "runs", "invalid-openrouter-model")), false); }); -test("report accepts populated accounting snapshots and preserves partial-pricing marker", async () => { - const project = tempProject(); +test("report accepts populated accounting snapshots and preserves partial-pricing marker", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeReportTopology(project); @@ -2188,8 +2211,8 @@ test("report accepts populated accounting snapshots and preserves partial-pricin assert.equal(fs.readFileSync(metadataPath, "utf8"), "{"); }); -test("report validates current artifacts without rewriting agent-owned bytes", async () => { - const project = tempProject(); +test("report validates current artifacts without rewriting agent-owned bytes", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-current-artifacts"); const reportDir = path.join(runData.run_root, "artifacts", "final-report"); const reportPath = path.join(reportDir, "report.json"); @@ -2221,8 +2244,8 @@ test("report validates current artifacts without rewriting agent-owned bytes", a assertFinalReportUnchanged(reportDir, reportSnapshot); }); -test("report displays authenticated final-verifier warnings while preserving the report files", async () => { - const project = tempProject(); +test("report displays authenticated final-verifier warnings while preserving the report files", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-host-warnings"); const reportDir = path.join(runData.run_root, "artifacts", "final-report"); fs.mkdirSync(reportDir, { recursive: true }); @@ -2249,8 +2272,8 @@ test("report displays authenticated final-verifier warnings while preserving the assertFinalReportUnchanged(reportDir, before); }); -test("report exposes terminal partial coverage and bundles only the authenticated runtime publication", async () => { - const project = tempProject(); +test("report exposes terminal partial coverage and bundles only the authenticated runtime publication", async (t) => { + const project = tempProject(t); const { run, report, stateBytes } = await createTerminalPartialReport(project, "report-runtime-partial"); const staleDirectory = path.join(run.run_root, "review", "runtime-report", "unverified-generation"); fs.mkdirSync(staleDirectory, { recursive: true }); @@ -2288,8 +2311,8 @@ test("report exposes terminal partial coverage and bundles only the authenticate assert.deepEqual(fs.readFileSync(path.join(run.run_root, "state.json")), stateBytes); }); -test("report verification is optional and unchecked bundles contain only the labeled report pair", async () => { - const project = tempProject(); +test("report verification is optional and unchecked bundles contain only the labeled report pair", async (t) => { + const project = tempProject(t); const { run, report, stateBytes } = await createTerminalPartialReport(project, "report-optional-verification"); const receipt = report.publications?.find((publication) => path.basename(publication.path) === "terminal.json"); assert.ok(receipt); @@ -2346,8 +2369,8 @@ test("report verification is optional and unchecked bundles contain only the lab assert.match(JSON.stringify(parseJson(withoutAccounting).diagnostics), /REPORT_ACCOUNTING_UNAVAILABLE/u); }); -test("report bundle --require-verified rejects receipt bytes changed only during archive capture", async () => { - const project = tempProject(); +test("report bundle --require-verified rejects receipt bytes changed only during archive capture", async (t) => { + const project = tempProject(t); const { run, report } = await createTerminalPartialReport(project, "report-runtime-receipt-change"); const receipt = report.publications?.find((publication) => path.basename(publication.path) === "terminal.json"); assert.ok(receipt); @@ -2362,8 +2385,8 @@ test("report bundle --require-verified rejects receipt bytes changed only during assert.equal(fs.existsSync(path.join(project, ".ultrafuzz", "bundles", `${run.run_id}-report-bundle.zip`)), false); }); -test("eval report validates the registered summary and never synthesizes missing Markdown", async () => { - const project = tempProject(); +test("eval report validates the registered summary and never synthesizes missing Markdown", async (t) => { + const project = tempProject(t); const evalRunId = "eval-report-strict"; const runRoot = path.join(project, ".ultrafuzz", "evals", "runs", evalRunId); const summaryPath = path.join(runRoot, "summary.json"); @@ -2451,8 +2474,8 @@ test("eval report validates the registered summary and never synthesizes missing assert.equal((parseJson(valid).data as { eval_run_id: string }).eval_run_id, evalRunId); }); -test("report --require-verified rejects final_severity compatibility aliases without rewriting artifacts", async () => { - const project = tempProject(); +test("report --require-verified rejects final_severity compatibility aliases without rewriting artifacts", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-rejects-severity-alias"); const reportDir = path.join(runData.run_root, "artifacts", "final-report"); const reportPath = path.join(reportDir, "report.json"); @@ -2474,8 +2497,8 @@ test("report --require-verified rejects final_severity compatibility aliases wit assertFinalReportUnchanged(reportDir, reportSnapshot); }); -test("report --require-verified does not synthesize missing Markdown", async () => { - const project = tempProject(); +test("report --require-verified does not synthesize missing Markdown", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-missing-markdown"); const reportDir = path.join(runData.run_root, "artifacts", "final-report"); const reportPath = path.join(reportDir, "report.json"); @@ -2497,8 +2520,8 @@ test("report --require-verified does not synthesize missing Markdown", async () assert.equal(digest(fs.readFileSync(reportPath)), jsonSha256Before); }); -test("report accepts canonical severity and complete proof without rewriting either artifact", async () => { - const project = tempProject(); +test("report accepts canonical severity and complete proof without rewriting either artifact", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-canonical-severity", writeBoundedDedupeReportTopology); const reportDir = path.join(runData.run_root, "artifacts", "final-report"); const reportPath = path.join(reportDir, "report.json"); @@ -2521,8 +2544,8 @@ test("report accepts canonical severity and complete proof without rewriting eit assert.equal(Object.hasOwn(report.issues[0] ?? {}, "final_severity"), false); }); -test("report --require-verified rejects malformed issues without rewriting stale bytes", async () => { - const project = tempProject(); +test("report --require-verified rejects malformed issues without rewriting stale bytes", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-current-malformed"); const reportDir = path.join(runData.run_root, "artifacts", "final-report"); const reportPath = path.join(reportDir, "report.json"); @@ -2557,8 +2580,8 @@ test("report --require-verified rejects malformed issues without rewriting stale assertFinalReportUnchanged(reportDir, reportSnapshot); }); -test("report --require-verified rejects legacy report versions without a compatibility reader", async () => { - const project = tempProject(); +test("report --require-verified rejects legacy report versions without a compatibility reader", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-legacy-version"); const reportDir = path.join(runData.run_root, "artifacts", "final-report"); const reportPath = path.join(reportDir, "report.json"); @@ -2580,8 +2603,8 @@ test("report --require-verified rejects legacy report versions without a compati assertFinalReportUnchanged(reportDir, reportSnapshot); }); -test("agent-owned bytes stay identical across validation, sync, aggregation, report, dashboard, eval, and bundle reads", async () => { - const project = tempProject(); +test("agent-owned bytes stay identical across validation, sync, aggregation, report, dashboard, and bundle reads", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeByteIdentityTopology(project); const env = fakeSmithersEnv(project, true, ["aggregate-test-files"]); @@ -2678,69 +2701,6 @@ test("agent-owned bytes stay identical across validation, sync, aggregation, rep } assertAgentBytesUnchanged(); - const uploads: EvalArtifactUpload[] = []; - const reporter: EvalReporter = { - name: "byte-identity", - async onPlan() {}, - async onRowStart() {}, - async onNodeEvent() {}, - async onArtifact(artifact) { - uploads.push(artifact); - }, - async onRowFinish() {}, - async onScores() {}, - async finalize() { - return {}; - } - }; - const row: EvalMatrixRow = { - id: "byte-identity-row", - target_id: "target", - variant_id: "variant", - trial_id: "trial", - run_id: runData.run_id, - target: { - id: "target", - repo: "https://example.com/target.git", - ref: "a".repeat(40), - ground_truth: "target.yml", - ground_truth_path: path.join(project, "target.yml"), - sensitivity: "public" - }, - variant: { id: "variant" }, - runner_model_profile: "runner", - judge_model_profile: "judge" - }; - const cursorPath = path.join(project, ".ultrafuzz", "evals", "byte-identity-cursor.json"); - fs.mkdirSync(path.dirname(cursorPath), { recursive: true }); - const telemetry = new NodeTelemetryPump({ - runRoot: runData.run_root, - row, - reporters: [reporter], - policy: { - node_telemetry: true, - heartbeat_interval_seconds: 60, - artifacts: { - mode: "upload", - mode_explicit: true, - include: ["report.json", "report.md"], - max_file_bytes: 1024 * 1024 - } - }, - cursorPath, - retryDelayMs: 0 - }); - const drained = await telemetry.drain(); - assert.deepEqual(drained.warnings, []); - assert.deepEqual(uploads.map((upload) => upload.relativePath).sort(), ["report.json", "report.md"]); - for (const upload of uploads) { - assert.ok(upload.read); - const expected = - upload.relativePath === "report.json" ? agentBytes.get(reportJsonPath) : agentBytes.get(reportMarkdownPath); - assert.deepEqual(await upload.read(), expected); - } - assertAgentBytesUnchanged(); - const bundled = await cli(project, ["report", "bundle", runData.run_id, "--json"]); assert.equal(bundled.code, 0, bundled.stderr); const bundlePath = (parseJson(bundled).data as { zip_path: string }).zip_path; @@ -2752,8 +2712,8 @@ test("agent-owned bytes stay identical across validation, sync, aggregation, rep assertAgentBytesUnchanged(); }); -test("report bundle creates a portable ZIP without workspaces", async () => { - const project = tempProject(); +test("report bundle creates a portable ZIP without workspaces", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeReportTopology(project); @@ -2956,8 +2916,8 @@ test("report bundle creates a portable ZIP without workspaces", async () => { assertFinalReportUnchanged(reportDir, reportSnapshot); }); -test("report bundle manifest records files it could not package", async () => { - const project = tempProject(); +test("report bundle manifest records files it could not package", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-bundle-omissions"); const engineLogDir = path.join(runData.run_root, "smithers", "logs"); @@ -3000,8 +2960,8 @@ test("report bundle manifest records files it could not package", async () => { ); }); -test("report bundle --require-verified rejects a changed authenticated publication from any finalized producer", async () => { - const project = tempProject(); +test("report bundle --require-verified rejects a changed authenticated publication from any finalized producer", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-bundle-mutated-publication"); const artifactDir = path.join(runData.run_root, "artifacts", "project-discovery"); const supportPublicationPath = "evidence/discovery-trace.txt"; @@ -3024,8 +2984,8 @@ test("report bundle --require-verified rejects a changed authenticated publicati ); }); -test("report bundle --require-verified rejects manifest bytes injected only into the recursive archive read", async () => { - const project = tempProject(); +test("report bundle --require-verified rejects manifest bytes injected only into the recursive archive read", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-bundle-manifest-read-injection"); const artifactDir = path.join(runData.run_root, "artifacts", "project-discovery"); fs.writeFileSync(path.join(artifactDir, "stdout.txt"), "generated stdout\n", "utf8"); @@ -3048,8 +3008,8 @@ test("report bundle --require-verified rejects manifest bytes injected only into ); }); -test("report bundle --require-verified rejects graph-fingerprint bytes injected only into the archive read", async () => { - const project = tempProject(); +test("report bundle --require-verified rejects graph-fingerprint bytes injected only into the archive read", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-bundle-fingerprint-read-injection"); const artifactDir = path.join(runData.run_root, "artifacts", "project-discovery"); fs.writeFileSync(path.join(artifactDir, "stdout.txt"), "generated stdout\n", "utf8"); @@ -3072,8 +3032,8 @@ test("report bundle --require-verified rejects graph-fingerprint bytes injected ); }); -test("report bundle --require-verified rejects a persistently tampered sealed graph fingerprint before writing a ZIP", async () => { - const project = tempProject(); +test("report bundle --require-verified rejects a persistently tampered sealed graph fingerprint before writing a ZIP", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-bundle-fingerprint-persistent-tamper"); const fingerprintPath = path.join(runData.run_root, "graph.fingerprint"); const tamperedBytes = Buffer.from(`${"a".repeat(64)}\n`, "utf8"); @@ -3090,8 +3050,8 @@ test("report bundle --require-verified rejects a persistently tampered sealed gr ); }); -test("report bundle --require-verified rejects a declared report producer that claims success without finalization authority", async () => { - const project = tempProject(); +test("report bundle --require-verified rejects a declared report producer that claims success without finalization authority", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-bundle-invalid-success-claim"); const layout = layoutForRunRoot(runData.run_root, runData.run_id); updateNodeState(layout, "final-report", { @@ -3113,8 +3073,8 @@ test("report bundle --require-verified rejects a declared report producer that c ); }); -test("report bundle --require-verified applies canonical report-pair validation to custom declarations", async () => { - const project = tempProject(); +test("report bundle --require-verified applies canonical report-pair validation to custom declarations", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeCustomReportTopology(project); const run = await cli( @@ -3148,8 +3108,8 @@ test("report bundle --require-verified applies canonical report-pair validation ); }); -test("report bundle preserves the exact validated event-record-v2 journal snapshot", async () => { - const project = tempProject(); +test("report bundle preserves the exact validated event-record-v2 journal snapshot", async (t) => { + const project = tempProject(t); const runId = "report-bundle-event-snapshot"; const runRoot = (await createReportRun(project, runId)).run_root; const record = bundleFixtureEvent(runId); @@ -3165,8 +3125,8 @@ test("report bundle preserves the exact validated event-record-v2 journal snapsh assert.deepEqual(captured, journalBytes); }); -test("report bundle treats only an absent event journal as optional", async () => { - const project = tempProject(); +test("report bundle treats only an absent event journal as optional", async (t) => { + const project = tempProject(t); const runId = "report-bundle-no-event-journal"; const runRoot = (await createReportRun(project, runId)).run_root; fs.rmSync(path.join(runRoot, "events.jsonl"), { force: true }); @@ -3180,8 +3140,8 @@ test("report bundle treats only an absent event journal as optional", async () = assert.equal(zip.readAsText("graph.fingerprint"), fs.readFileSync(path.join(runRoot, "graph.fingerprint"), "utf8")); }); -test("report bundle --require-verified rejects historical runs without current sealed workflow authority", async () => { - const project = tempProject(); +test("report bundle --require-verified rejects historical runs without current sealed workflow authority", async (t) => { + const project = tempProject(t); assert.equal((await cli(project, ["init", "--force"])).code, 0); const runId = "report-bundle-historical-unsealed"; const runRoot = path.join(project, ".ultrafuzz", "runs", runId); @@ -3196,7 +3156,7 @@ test("report bundle --require-verified rejects historical runs without current s }); test("report bundle --require-verified fails closed on every present invalid event journal", async (context) => { - const project = tempProject(); + const project = tempProject(context); const cases: Array<{ name: string; prepare(eventsPath: string, runId: string): void; @@ -3274,8 +3234,8 @@ test("report bundle --require-verified fails closed on every present invalid eve } }); -test("report bundle packages incomplete runs without a final-report JSON", async () => { - const project = tempProject(); +test("report bundle packages incomplete runs without a final-report JSON", async (t) => { + const project = tempProject(t); const runData = await createReportRun(project, "report-bundle-incomplete"); const bundled = await cli(project, ["report", "bundle", runData.run_id, "--json"]); @@ -3295,8 +3255,8 @@ test("report bundle packages incomplete runs without a final-report JSON", async ); }); -test("stats reads strict local current evidence and reports genuinely absent ledgers", async () => { - const project = tempProject(); +test("stats reads strict local current evidence and reports genuinely absent ledgers", async (t) => { + const project = tempProject(t); const fixture = writeStatsFixture(project, "stats-local", { linked: false }); const captured = await cli(project, ["stats", fixture.runId, "--json"]); @@ -3347,7 +3307,7 @@ test("stats reads strict local current evidence and reports genuinely absent led }); test("stats falls back to unchanged local evidence when workflow event output is malformed", async (context) => { - const project = tempProject(); + const project = tempProject(context); assert.equal((await cli(project, ["init", "--force"])).code, 0); writeSmallTopology(project); const env = fakeSmithersEnv(project); @@ -3381,8 +3341,8 @@ test("stats falls back to unchanged local evidence when workflow event output is } }); -test("stats queries a current report bundle offline and reports genuinely missing ledgers", async () => { - const project = tempProject(); +test("stats queries a current report bundle offline and reports genuinely missing ledgers", async (t) => { + const project = tempProject(t); const fixture = writeStatsFixture(project, "stats-bundle"); const bundlePath = path.join(project, "stats-bundle.zip"); writeStatsBundle(bundlePath, fixture); @@ -3418,8 +3378,8 @@ test("stats queries a current report bundle offline and reports genuinely missin ); }); -test("stats rejects a historical report-bundle manifest", async () => { - const project = tempProject(); +test("stats rejects a historical report-bundle manifest", async (t) => { + const project = tempProject(t); const fixture = writeStatsFixture(project, "stats-v1-bundle"); const bundlePath = path.join(project, "stats-v1-bundle.zip"); writeStatsBundle(bundlePath, fixture, { @@ -3435,7 +3395,7 @@ test("stats rejects a historical report-bundle manifest", async () => { }); test("stats fails closed for malformed present bundle evidence", async (context) => { - const project = tempProject(); + const project = tempProject(context); const cases: Array<{ name: string; member?: string; @@ -3609,7 +3569,7 @@ test("stats fails closed for malformed present bundle evidence", async (context) }); test("stats binds the v3 manifest entry count and rejects ZIP path aliases", async (context) => { - const project = tempProject(); + const project = tempProject(context); await context.test("manifest entry count", async () => { const fixture = writeStatsFixture(project, "stats-count-mismatch"); const bundlePath = path.join(project, "stats-count-mismatch.zip"); @@ -3670,8 +3630,8 @@ test("stats binds the v3 manifest entry count and rejects ZIP path aliases", asy }); }); -test("stats rejects local evidence whose leaf is a symlink", async () => { - const project = tempProject(); +test("stats rejects local evidence whose leaf is a symlink", async (t) => { + const project = tempProject(t); const fixture = writeStatsFixture(project, "stats-symlink"); const usagePath = path.join(fixture.runRoot, "usage.jsonl"); const outsidePath = path.join(project, "outside-usage.jsonl"); @@ -3684,8 +3644,8 @@ test("stats rejects local evidence whose leaf is a symlink", async () => { assert.match(JSON.stringify(parseJson(captured).diagnostics), /symlink/u); }); -test("stats rejects linked local evidence without sealed workflow authority", async () => { - const project = tempProject(); +test("stats rejects linked local evidence without sealed workflow authority", async (t) => { + const project = tempProject(t); const fixture = writeStatsFixture(project, "stats-missing-control-authority"); const captured = await cli(project, ["stats", fixture.runId, "--json"]); diff --git a/packages/cli/test/e2e/campaign-resume.test.ts b/packages/cli/test/e2e/campaign-resume.test.ts new file mode 100644 index 000000000..06d9b6a43 --- /dev/null +++ b/packages/cli/test/e2e/campaign-resume.test.ts @@ -0,0 +1,523 @@ +import assert from "node:assert/strict"; +import { execFile, execFileSync } from "node:child_process"; +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; +import { setTimeout as delay } from "node:timers/promises"; +import { fileURLToPath } from "node:url"; + +// One campaign through the shipped product path: `init`, `run`, `resume`, `status`, `stats`, +// `report`, and `events` each run as their own `ultrafuzz` process, and the generated workflow runs +// on the pinned Smithers engine that `run` and `resume` install from npm and start under Bun. Only +// the model is fake: a stub `codex` executable on PATH. Other runtime and CLI tests drive a fake +// `smithers` shell script, or run the engine on hand-written workflows. + +const CLI_ENTRYPOINT = fileURLToPath(new URL("../../../dist/index.js", import.meta.url)); +const MINUTE = 60_000; +const AGENT_NODES = ["project-discovery", "summarize", "final-report"] as const; +const INTERRUPTED_NODE = "summarize"; + +const TOPOLOGY = `version: 2 +defaults: + strategy_loops: 1 +nodes: + - id: __start__ + kind: meta + role: start + depends_on: [] + - id: project-discovery + kind: agentic + prompt: setup/project-discovery.md + depends_on: + - __start__ + outputs: + - path: discovery.txt + contract: ultrafuzz/text@1 + primary: true + - id: summarize + kind: agentic + prompt: setup/project-discovery.md + depends_on: + - project-discovery + outputs: + - path: summary.txt + contract: ultrafuzz/text@1 + primary: true + - id: final-report + kind: agentic + prompt: review/final-report.md + depends_on: + - summarize + outputs: + - path: report.md + contract: ultrafuzz/nonempty-markdown@1 + primary: true + - path: report.json + contract: ultrafuzz/report@3 + - id: __finish__ + kind: meta + role: finish + depends_on: + - final-report +`; + +// A public campaign needs no disclosure acknowledgement. CodexAgent routes to model:openai. +const PUBLIC_DATA_GOVERNANCE_POLICY = { + schema_version: "ultrafuzz.data-governance-policy.v1", + sensitivity: "public", + source_destinations: ["model:openai"], + artifact_destinations: [], + destination_policies: [ + { + destination: "model:openai", + processor: "stub codex", + region: "local", + retention_policy: "test fixture", + training_policy: "none", + dpa_status: "not-required", + minimization_policy: "synthetic fixture only", + data_handling_basis: "public test fixture" + } + ], + openrouter_model_allowlist: [] +}; + +interface StubConfig { + logPath: string; + holdPath: string; + holdNode: string; +} + +interface AgentCall { + node: string; + pid: number; + event: "started" | "held" | "completed" | "failed"; + error?: string; +} + +/** + * The fake `codex` binary. The engine spawns it exactly as it spawns Codex (`codex exec ... --json + * -`, prompt on stdin). It writes the `text@1` and `report@3` outputs that the prompt's output + * contract names, renders the final report's markdown with the `ultrafuzz report render` command + * the prompt gives, and prints the Codex JSONL the engine parses. While `holdPath` exists, it holds + * `holdNode` open for up to 10 minutes, so the test can kill the controller in the middle of it. It + * fails, and logs why, when the prompt no longer has the shape it parses. The function is + * serialized into the binary, so it may use only globals. + */ +function stubCodex(config: StubConfig): void { + const fs = process.getBuiltinModule("node:fs"); + const path = process.getBuiltinModule("node:path"); + const { execFileSync } = process.getBuiltinModule("node:child_process"); + const args = process.argv.slice(2); + const prompt = fs.readFileSync(0, "utf8"); + const node = (process.env.SMITHERS_NODE_ID ?? "").replace(/^node:/u, ""); + const log = (event: AgentCall["event"], error?: string): void => + fs.appendFileSync(config.logPath, `${JSON.stringify({ node, pid: process.pid, event, error })}\n`); + log("started"); + if (node === config.holdNode && fs.existsSync(config.holdPath)) { + log("held"); + setTimeout(() => process.exit(1), 10 * 60_000); + return; + } + try { + const outputs = new Map(); + for (const [, file, contract] of prompt.matchAll(/^- Path: `([^`]+)`.*\n\s+Contract: `([^`]+)`/gmu)) { + if (file === undefined || contract === undefined) continue; + if (!["ultrafuzz/text@1", "ultrafuzz/report@3", "ultrafuzz/nonempty-markdown@1"].includes(contract)) { + throw new Error(`stub codex cannot write ${contract}`); + } + fs.mkdirSync(path.dirname(file), { recursive: true }); + outputs.set(contract, file); + } + if (outputs.size === 0) throw new Error("stub codex found no output contract in the prompt"); + const text = outputs.get("ultrafuzz/text@1"); + if (text !== undefined) fs.writeFileSync(text, `stub output for ${node}\n`); + const report = outputs.get("ultrafuzz/report@3"); + if (report !== undefined) { + // Copy the host-injected authorities the prompt names, as the final-report prompt instructs. + const authority = (suffix: string): Record => { + const file = [...prompt.matchAll(/workspace-relative file "([^"]+)"/gu)] + .map((match) => match[1]) + .find((candidate) => candidate?.endsWith(suffix) === true); + if (file === undefined) throw new Error(`stub codex found no ${suffix} authority in the prompt`); + return JSON.parse(fs.readFileSync(path.resolve(file), "utf8")) as Record; + }; + const data = authority(".final-report-prompt.json"); + const document = { + schema_version: "ultrafuzz.report.v3", + run_metadata: { ...authority(".final-report-run-metadata.json"), agent_execution: data.agent_execution }, + issues: [], + non_production_outcomes: [], + property_provenance: [], + property_implementation_coverage: data.property_implementation_coverage + }; + fs.writeFileSync(report, `${JSON.stringify(document, null, 2)}\n`); + const render = /^ultrafuzz report render (.+)$/mu.exec(prompt)?.[1]; + // Every argument must be a `--flag 'value'` pair, so none is silently dropped. + if (render === undefined || !/^(?:--[a-z-]+ '[^']+' ?)+$/u.test(render)) { + throw new Error(`stub codex cannot parse the report render command: ${render}`); + } + const renderArgs = [...render.matchAll(/(--[a-z-]+) '([^']+)'/gu)].flatMap(([, flag, value]) => [ + flag ?? "", + value ?? "" + ]); + // The renderer writes report.md; its stdout must not interleave with the JSONL below. + execFileSync("ultrafuzz", ["report", "render", ...renderArgs], { stdio: ["ignore", "ignore", "inherit"] }); + } + } catch (error) { + log("failed", String(error)); + throw error; + } + const emit = (event: object): boolean => process.stdout.write(`${JSON.stringify(event)}\n`); + emit({ type: "thread.started", thread_id: `stub-${node}` }); + emit({ type: "turn.started" }); + emit({ type: "item.completed", item: { id: "answer", type: "agent_message", text: `stub completed ${node}` } }); + const lastMessage = args.indexOf("--output-last-message"); + if (lastMessage >= 0) fs.writeFileSync(args[lastMessage + 1] ?? "", `stub completed ${node}`); + emit({ type: "turn.completed", usage: { input_tokens: 100, cached_input_tokens: 0, output_tokens: 10 } }); + log("completed"); +} + +interface Campaign { + root: string; + project: string; + runId: string; + env: NodeJS.ProcessEnv; + logPath: string; + holdPath: string; +} + +interface HealthValue { + status: string; + verdict: string; + ended: boolean; + progress: { finished: number; in_progress: number; pending: number; failed: number; skipped: number; total: number }; + model_mix: Array<{ attempts: number }>; + report: { status: string; completion: string; verification: string }; +} + +interface StatsValue { + status: string; + nodes: Array<{ + node_id: string; + status: string; + attempt_count: number | null; + executed_attempt_count: number | null; + }>; + totals: { node_count: number; status_counts: Record }; +} + +interface WorkflowEvent { + sequence: number; + category: string; + node_id: string | null; +} + +function prepareCampaign(): Campaign { + const root = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-e2e-")); + const project = path.join(root, "target"); + const bin = path.join(root, "bin"); + const codexHome = path.join(root, "codex-home"); + const tmp = path.join(root, "tmp"); + for (const directory of [project, bin, codexHome, tmp]) fs.mkdirSync(directory, { mode: 0o700 }); + const campaign: Campaign = { + root, + project, + runId: `e2e-${process.pid}-${Date.now().toString(36)}`, + logPath: path.join(root, "agent-calls.jsonl"), + holdPath: path.join(root, `hold-${INTERRUPTED_NODE}`), + env: { + ...Object.fromEntries( + Object.entries(process.env).filter( + ([name]) => name !== "NODE_TEST_CONTEXT" && !/^(?:ULTRAFUZZ|SMITHERS|CODEX|OPENAI)_/u.test(name) + ) + ), + PATH: `${bin}${path.delimiter}${process.env.PATH ?? ""}`, + // Subscription auth reads auth.json here, so the engine's agent preflight stays offline. + CODEX_HOME: codexHome, + // Keeps the operator controller that each engine command installs inside the fixture root. + TMPDIR: tmp, + ULTRAFUZZ_DATA_GOVERNANCE_POLICY: JSON.stringify(PUBLIC_DATA_GOVERNANCE_POLICY), + ULTRAFUZZ_PRICING_CATALOG_URL: "off" + } + }; + fs.writeFileSync( + path.join(codexHome, "auth.json"), + `${JSON.stringify({ auth_mode: "chatgpt", tokens: { access_token: "stub" } })}\n` + ); + const stubConfig: StubConfig = { logPath: campaign.logPath, holdPath: campaign.holdPath, holdNode: INTERRUPTED_NODE }; + const stub = `#!/usr/bin/env node\n(${String(stubCodex)})(${JSON.stringify(stubConfig)});\n`; + fs.writeFileSync(path.join(bin, "codex"), stub, { mode: 0o755 }); + fs.writeFileSync(path.join(project, "README.md"), "# Target\n"); + for (const args of [ + ["init", "-q"], + ["add", "README.md"], + ["commit", "-q", "-m", "fixture"] + ]) { + execFileSync("git", ["-c", "user.name=Ultrafuzz", "-c", "user.email=e2e@ultrafuzz.invalid", ...args], { + cwd: project, + stdio: "ignore" + }); + } + return campaign; +} + +async function ultrafuzz(campaign: Campaign, args: string[], timeoutMs = 5 * MINUTE): Promise { + const command = `ultrafuzz ${args.join(" ")}`; + const { error, stdout, stderr } = await new Promise<{ error: Error | null; stdout: string; stderr: string }>( + (resolve) => { + execFile( + process.execPath, + [CLI_ENTRYPOINT, ...args, "--project", campaign.project, "--json"], + { cwd: campaign.project, env: campaign.env, timeout: timeoutMs, killSignal: "SIGKILL", maxBuffer: 64 << 20 }, + // A failed envelope exits non-zero; the envelope itself is the evidence. + (failure, out, err) => resolve({ error: failure, stdout: out, stderr: err }) + ); + } + ); + let envelope: { ok: boolean; data: unknown; diagnostics: unknown[] }; + try { + envelope = JSON.parse(stdout) as typeof envelope; + } catch { + assert.fail(`${command} printed no JSON envelope (${error?.message ?? "exit 0"})\nstderr: ${stderr}`); + } + assert.equal(envelope.ok, true, `${command}: ${JSON.stringify(envelope.diagnostics)}`); + return envelope.data as T; +} + +async function waitFor(label: string, timeoutMs: number, probe: () => Promise | T | undefined) { + const deadline = Date.now() + timeoutMs; + for (;;) { + const value = await probe(); + if (value !== undefined) return value; + if (Date.now() > deadline) assert.fail(`timed out after ${timeoutMs / MINUTE} minutes waiting for ${label}`); + await delay(1_000); + } +} + +function agentCalls(campaign: Campaign): AgentCall[] { + if (!fs.existsSync(campaign.logPath)) return []; + // The last element is empty or a line the stub is still appending. + const lines = fs.readFileSync(campaign.logPath, "utf8").split("\n").slice(0, -1); + return lines.map((line) => JSON.parse(line) as AgentCall); +} + +/** A Linux process's command line; empty for a zombie, and undefined once it has been reaped. */ +function commandLine(pid: number | string): string | undefined { + try { + return fs.readFileSync(`/proc/${pid}/cmdline`, "utf8"); + } catch { + return undefined; + } +} + +/** Linux PIDs whose command line mentions `needle`. */ +function processesMentioning(needle: string): number[] { + return fs.readdirSync("/proc").flatMap((entry) => { + if (!/^\d+$/u.test(entry) || Number(entry) === process.pid) return []; + return commandLine(entry)?.includes(needle) === true ? [Number(entry)] : []; + }); +} + +function kill(pid: number): void { + try { + process.kill(pid, "SIGKILL"); + } catch { + // Already gone. + } +} + +function removeTree(root: string): void { + // Sealed execution snapshots are read-only directories; a leaked fixture must not fail the test. + const restore = (directory: string): void => { + fs.chmodSync(directory, 0o700); + for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { + if (entry.isDirectory()) restore(path.join(directory, entry.name)); + } + }; + try { + restore(root); + fs.rmSync(root, { recursive: true, force: true }); + } catch { + // Best effort. + } +} + +/** A task the engine finished must not start again: resume reuses it. */ +function assertNoFinishedTaskRestarted(events: WorkflowEvent[]): void { + const finished = new Set(); + for (const event of events) { + if (event.node_id === null) continue; + if (event.category === "NodeStarted") { + assert.equal(finished.has(event.node_id), false, `${event.node_id} started again at event ${event.sequence}`); + } + if (event.category === "NodeFinished") finished.add(event.node_id); + } + assert.ok(finished.size > 0, "the workflow event stream has no finished tasks"); +} + +/** Launches the campaign and SIGKILLs its detached controller while `INTERRUPTED_NODE` runs. */ +async function interruptMidRun(campaign: Campaign, mark: (phase: string) => void): Promise { + await ultrafuzz(campaign, ["init"], 2 * MINUTE); + const configPath = path.join(campaign.project, "ultrafuzz.toml"); + const generated = fs.readFileSync(configPath, "utf8"); + // init configures Codex API-key auth, whose preflight calls api.openai.com. + const config = generated.replace( + /^\[agents\.CodexAgent\]\n(?:[^[\n].*\n)*/mu, + '[agents.CodexAgent]\nauth = "subscription"\n' + ); + assert.notEqual(config, generated, "init no longer writes an [agents.CodexAgent] table"); + fs.writeFileSync(configPath, config); + fs.writeFileSync(path.join(campaign.project, ".ultrafuzz", "topology.yml"), TOPOLOGY); + fs.writeFileSync(campaign.holdPath, ""); + + const launched = await ultrafuzz<{ status: string; workflow_ids: string[] }>( + campaign, + ["run", "--run-id", campaign.runId], + 20 * MINUTE + ); + mark("run submitted"); + assert.equal(launched.status, "running"); + const [workflowRunId] = launched.workflow_ids; + assert.ok(workflowRunId !== undefined, "run did not report its workflow run ID"); + + const held = await waitFor(`${INTERRUPTED_NODE} to start`, 15 * MINUTE, () => { + const calls = agentCalls(campaign); + const call = calls.find((entry) => entry.node === INTERRUPTED_NODE && entry.event === "held"); + if (call === undefined && processesMentioning(workflowRunId).length === 0) { + assert.fail(`the workflow stopped before ${INTERRUPTED_NODE} started; agent calls: ${JSON.stringify(calls)}`); + } + return call; + }); + // A host crash takes down the detached engine and the supervisor that would otherwise restart it. + const controller = processesMentioning(workflowRunId); + assert.ok(controller.length > 0, "no detached controller process is running the workflow"); + for (const pid of controller) kill(pid); + // The held agent is reparented once the engine dies; a zombie counts as exited. + await waitFor("the controller and its agent to exit", MINUTE, () => + processesMentioning(workflowRunId).length === 0 && (commandLine(held.pid) ?? "") === "" ? true : undefined + ); + mark("controller killed"); +} + +test( + "a campaign whose controller is SIGKILLed mid-node resumes on the pinned engine without re-running finished nodes", + { timeout: 45 * MINUTE, skip: process.platform === "linux" ? false : "finds the detached controller through /proc" }, + async (t) => { + const campaign = prepareCampaign(); + const { runId } = campaign; + const cleanUp = (): void => { + // The supervisor's command line names only the run ID; everything else names the fixture root. + for (const pid of [...processesMentioning(campaign.root), ...processesMentioning(runId)]) kill(pid); + removeTree(campaign.root); + }; + // The campaign runs detached, so an interrupted test must stop it before dying of the signal. + const interrupted = (name: NodeJS.Signals): void => { + cleanUp(); + process.kill(process.pid, name); + }; + process.once("SIGINT", interrupted); + process.once("SIGTERM", interrupted); + process.once("exit", cleanUp); + // Phase timings in the test output show where CI time goes. + const started = Date.now(); + const mark = (phase: string): void => t.diagnostic(`${phase} after ${Math.round((Date.now() - started) / 1000)} s`); + try { + await interruptMidRun(campaign, mark); + + // Until the dead engine's heartbeat lease lapses, the engine still reports the run as + // running, and resume leaves a running run alone (`submitted: false`). + await waitFor("status to report the run orphaned", 5 * MINUTE, async () => { + const health = await ultrafuzz(campaign, ["status", runId]); + assert.equal(health.ended, false); + return health.verdict === "orphaned" ? health : undefined; + }); + fs.rmSync(campaign.holdPath); + const resumed = await ultrafuzz<{ submitted: boolean }>(campaign, ["resume", runId], 15 * MINUTE); + mark("resume submitted"); + assert.equal(resumed.submitted, true); + + // `events` reads the engine's event log without synchronizing the run, so it is the cheaper poll. + const ended = ["RunFinished", "RunFailed", "RunCancelled"]; + const events = await waitFor("the resumed workflow to end", 15 * MINUTE, async () => { + const current = await ultrafuzz<{ events: WorkflowEvent[]; truncated: boolean }>(campaign, ["events", runId]); + return current.events.some((event) => ended.includes(event.category)) ? current : undefined; + }); + mark("run ended"); + const calls = agentCalls(campaign); + assert.equal( + events.events.find((event) => ended.includes(event.category))?.category, + "RunFinished", + `agent calls: ${JSON.stringify(calls)}` + ); + // Every agent ran once, except the node the kill interrupted, which ran again after resume. + const starts = calls.filter((call) => call.event === "started"); + assert.deepEqual( + AGENT_NODES.map((node) => [node, starts.filter((call) => call.node === node).length]), + AGENT_NODES.map((node) => [node, node === INTERRUPTED_NODE ? 2 : 1]) + ); + assert.equal(events.truncated, false); + assert.equal(events.events.filter((event) => event.category === "RunStarted").length, 2); + assertNoFinishedTaskRestarted(events.events); + + const health = await ultrafuzz(campaign, ["status", runId]); + assert.equal(health.ended, true); + assert.equal(health.status, "succeeded"); + assert.deepEqual( + { + status: health.report.status, + completion: health.report.completion, + verification: health.report.verification + }, + { status: "available", completion: "complete", verification: "verified" } + ); + + const report = await ultrafuzz<{ source: string; json_path: string; markdown_path: string }>(campaign, [ + "report", + runId + ]); + assert.equal(report.source, "verified-runtime-report"); + const reportJson = JSON.parse(fs.readFileSync(report.json_path, "utf8")) as { run_metadata: { run_id: string } }; + assert.equal(reportJson.run_metadata.run_id, runId); + assert.match(fs.readFileSync(report.markdown_path, "utf8"), /\S/u); + + // `status` counts engine tasks and `stats` counts topology nodes; both must describe the same + // complete run. + const { finished, in_progress, pending, failed, skipped } = health.progress; + assert.deepEqual( + { finished, in_progress, pending, failed, skipped }, + { finished: health.progress.total, in_progress: 0, pending: 0, failed: 0, skipped: 0 } + ); + const stats = await ultrafuzz(campaign, ["stats", runId]); + assert.equal(stats.status, "succeeded"); + assert.deepEqual( + stats.nodes.map((node) => [node.node_id, node.status]), + AGENT_NODES.map((node) => [node, "succeeded"]) + ); + assert.equal(stats.totals.node_count, AGENT_NODES.length); + assert.equal(stats.totals.status_counts.succeeded, AGENT_NODES.length); + assert.equal(stats.nodes.find((node) => node.node_id === "project-discovery")?.executed_attempt_count, 1); + // status counts every agent invocation, including the one the kill interrupted. + const statusAttempts = health.model_mix.reduce((total, entry) => total + entry.attempts, 0); + assert.equal(statusAttempts, starts.length); + mark("checks done"); + + await t.test( + "stats counts the agent attempt the controller crash interrupted", + { + todo: "Smithers emits no terminal event for the attempt it abandons when the resumed run starts, so attempts.jsonl never records it (#1187)" + }, + () => { + const statsAttempts = stats.nodes.reduce((total, node) => total + (node.attempt_count ?? 0), 0); + assert.equal(statsAttempts, statusAttempts); + } + ); + } finally { + process.removeListener("SIGINT", interrupted); + process.removeListener("SIGTERM", interrupted); + process.removeListener("exit", cleanUp); + cleanUp(); + } + } +); diff --git a/packages/cli/test/eval-history.test.ts b/packages/cli/test/eval-history.test.ts index 24de48228..18d089acd 100644 --- a/packages/cli/test/eval-history.test.ts +++ b/packages/cli/test/eval-history.test.ts @@ -4,67 +4,10 @@ import os from "node:os"; import path from "node:path"; import test from "node:test"; -import { - EVAL_HISTORY_OBSERVATION_SCHEMA_VERSION, - emptyEvalHistory, - mergeEvalHistory, - type EvalHistoryObservation -} from "@ultrafuzz/evals"; +import { emptyEvalHistory } from "@ultrafuzz/evals"; import { runCli } from "../src/index.js"; -const FROZEN_RUN_TIMESTAMP = "2026-07-19T00:00:00.000Z"; -const TARGET_REVISION = "2222222222222222222222222222222222222222"; - -function frozenObservation(): EvalHistoryObservation { - return { - schema_version: EVAL_HISTORY_OBSERVATION_SCHEMA_VERSION, - id: "run-1:target-a:baseline:benchmark-smoke", - benchmark: "ultrafuzz-bench", - lane: "smoke", - status: "succeeded", - target: "target-a", - variant: "baseline", - trial_count: 1, - run_timestamp: FROZEN_RUN_TIMESTAMP, - candidate_commit: "1111111111111111111111111111111111111111", - candidate_repository_url: "https://github.com/monad-developers/ultrafuzz", - cohort_fingerprint: `sha256:${"a".repeat(64)}`, - target_revisions: [{ target: "target-a", revision: TARGET_REVISION }], - model_profile: "benchmark-smoke", - model: "gpt-5.6-luna", - reasoning_effort: "high", - execution_policy_fingerprint: `sha256:${"d".repeat(64)}`, - scoring_fingerprint: `sha256:${"b".repeat(64)}`, - precision: 0.75, - recall: 0.5, - f1: 0.6, - cumulative_unique_true_positives: 2, - ground_truth_bug_count: 2, - wall_clock_seconds: 120, - wall_clock_completeness: { status: "complete", reasons: [] }, - cost_usd: 1.25, - cost_completeness: { status: "complete", reasons: [] }, - executed_case_count: 1, - graded_case_count: 1, - publication_url: "https://github.com/monad-developers/ultrafuzz/actions/runs/123/artifacts", - target_publication: { - target: "target-a", - repository: "https://example.com/target-a", - revision: TARGET_REVISION, - status: "succeeded", - executed_case_count: 1, - graded_case_count: 1, - publication_location: { - bundle_path: "public-results.json", - report_paths: ["reports/target-a-baseline-trial-1/report.md", "reports/target-a-baseline-trial-1/report.json"] - } - }, - source_eval_run_id: "run-1", - source_artifact: "https://github.com/monad-developers/ultrafuzz/actions/runs/1" - }; -} - test("eval history renders and checks deterministic public charts", async () => { const project = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-cli-history-")); fs.mkdirSync(path.join(project, "benchmarks", "ultrafuzzbench"), { recursive: true }); @@ -81,131 +24,20 @@ test("eval history renders and checks deterministic public charts", async () => assert.equal(result.ok, true); assert.equal(result.data.observations, 0); assert.deepEqual(fs.readdirSync(path.join(project, "docs", "assets", "eval-history")).sort(), [ - "cost.svg", - "cumulative-unique-true-positives.svg", - "f1.svg", "latest-summary.svg", "performance-cost.svg", - "precision.svg", - "quality.svg", - "recall.svg", - "wall-clock-time.svg" + "quality.svg" ]); const checked = await invoke(project, ["eval", "history", "--project", project, "--check", "--json"]); assert.equal(checked.code, 0, checked.stderr || checked.stdout); - const aged = await invoke(project, [ - "eval", - "history", - "--project", - project, - "--check", - "--max-age-days", - "7", - "--json" - ]); - assert.equal(aged.code, 1); - assert.equal( - (JSON.parse(aged.stdout) as { diagnostics: Array<{ message: string }> }).diagnostics[0]?.message, - "eval history has no observation to age against the requested 7 day maximum" - ); - - fs.appendFileSync(path.join(project, "docs", "assets", "eval-history", "precision.svg"), "stale\n"); + fs.appendFileSync(path.join(project, "docs", "assets", "eval-history", "quality.svg"), "stale\n"); const stale = await invoke(project, ["eval", "history", "--project", project, "--check", "--json"]); assert.equal(stale.code, 1); assert.equal((JSON.parse(stale.stdout) as { ok: boolean }).ok, false); }); -test("eval history asserts a requested maximum observation age without changing the default check", async () => { - const project = fs.mkdtempSync(path.join(os.tmpdir(), "ufz-cli-history-age-")); - fs.mkdirSync(path.join(project, "benchmarks", "ultrafuzzbench"), { recursive: true }); - fs.writeFileSync( - path.join(project, "benchmarks", "ultrafuzzbench", "history.json"), - `${JSON.stringify(mergeEvalHistory(emptyEvalHistory(), [frozenObservation()]))}\n`, - "utf8" - ); - - const rendered = await invoke(project, ["eval", "history", "--project", project, "--json"]); - assert.equal(rendered.code, 0, rendered.stderr || rendered.stdout); - const checked = await invoke(project, ["eval", "history", "--project", project, "--check", "--json"]); - assert.equal(checked.code, 0, checked.stderr || checked.stdout); - const generous = await invoke(project, [ - "eval", - "history", - "--project", - project, - "--check", - "--max-age-days", - "100000", - "--json" - ]); - assert.equal(generous.code, 0, generous.stderr || generous.stdout); - assert.equal(generous.stdout, checked.stdout); - - const stale = await invoke(project, [ - "eval", - "history", - "--project", - project, - "--check", - "--max-age-days", - "1", - "--json" - ]); - assert.equal(stale.code, 1); - const failure = JSON.parse(stale.stdout) as { - ok: boolean; - diagnostics: Array<{ code: string; message: string }>; - }; - assert.equal(failure.ok, false); - assert.match( - failure.diagnostics[0]?.message ?? "", - new RegExp( - `^newest eval history observation ran at ${FROZEN_RUN_TIMESTAMP.replace(/\./u, "\\.")}, [0-9.]+ days ago, exceeding the requested 1 day maximum$`, - "u" - ) - ); - - const appended = await invoke(project, ["eval", "history", "run-1", "--project", project, "--max-age-days", "1"]); - assert.equal(appended.code, 1); - assert.match(appended.stderr, /--max-age-days cannot append an eval run/u); -}); - -test("eval history fails a requested maximum age when the newest observation is dated in the future", async () => { - const project = fs.mkdtempSync(path.join(os.tmpdir(), "ufz-cli-history-future-")); - fs.mkdirSync(path.join(project, "benchmarks", "ultrafuzzbench"), { recursive: true }); - const future = "2099-01-01T00:00:00.000Z"; - fs.writeFileSync( - path.join(project, "benchmarks", "ultrafuzzbench", "history.json"), - `${JSON.stringify(mergeEvalHistory(emptyEvalHistory(), [{ ...frozenObservation(), run_timestamp: future }]))}\n`, - "utf8" - ); - - const rendered = await invoke(project, ["eval", "history", "--project", project, "--json"]); - assert.equal(rendered.code, 0, rendered.stderr || rendered.stdout); - const dated = await invoke(project, [ - "eval", - "history", - "--project", - project, - "--check", - "--max-age-days", - "2", - "--json" - ]); - assert.equal(dated.code, 1); - const failure = JSON.parse(dated.stdout) as { ok: boolean; diagnostics: Array<{ message: string }> }; - assert.equal(failure.ok, false); - assert.match( - failure.diagnostics[0]?.message ?? "", - new RegExp( - `^newest eval history observation ran at ${future.replace(/\./u, "\\.")}, which is in the future at [0-9TZ:.-]+, so its age cannot be measured against the requested 2 day maximum$`, - "u" - ) - ); -}); - async function invoke(project: string, argv: string[]): Promise<{ code: number; stdout: string; stderr: string }> { let stdout = ""; let stderr = ""; diff --git a/packages/cli/test/json-validate.test.ts b/packages/cli/test/json-validate.test.ts index c86fad21f..bddde4f09 100644 --- a/packages/cli/test/json-validate.test.ts +++ b/packages/cli/test/json-validate.test.ts @@ -38,7 +38,7 @@ import { evmbenchSchemaBundleDigest, evmbenchSchemaDirectory } from "@ultrafuzz/evmbench"; -import { EVAL_PUBLICATION_STATE_SCHEMA_ID, evalSchemaBundleDigest, evalSchemaDirectory } from "@ultrafuzz/evals"; +import { EVAL_GROUND_TRUTH_SCHEMA_ID, evalSchemaBundleDigest, evalSchemaDirectory } from "@ultrafuzz/evals"; import { MAX_PUBLIC_BENCHMARK_BUNDLE_BYTES, MODAL_NODE_INPUT_SCHEMA_ID, @@ -665,19 +665,15 @@ test("json validate recognizes the pinned resolved-config schema and reports the test("json validate recognizes the pinned eval schema and reports the owning eval bundle", async () => { const temporary = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ultrafuzz-json-eval-")); try { - const schema = path.join(evalSchemaDirectory(), "eval-publication-state.schema.json"); - const publicationState = path.join(temporary, "publication-state.json"); + const schema = path.join(evalSchemaDirectory(), "eval-ground-truth.schema.json"); + const groundTruth = path.join(temporary, "ground-truth.json"); fs.writeFileSync( - publicationState, - `${JSON.stringify({ - schema_version: "ultrafuzz.eval.publication.v1", - status: "publishable", - diagnostics: [] - })}\n`, + groundTruth, + `${JSON.stringify({ schema_version: "ultrafuzz.eval-ground-truth.v1", bugs: [] })}\n`, "utf8" ); - const validCapture = await capture(["json", "validate", "--schema", schema, "--file", publicationState, "--json"]); + const validCapture = await capture(["json", "validate", "--schema", schema, "--file", groundTruth, "--json"]); assert.equal(validCapture.code, 0); const envelope = JSON.parse(validCapture.stdout) as { ok: boolean; @@ -689,10 +685,10 @@ test("json validate recognizes the pinned eval schema and reports the owning eva assert.equal(envelope.ok, true); assert.equal(envelope.data.status, "valid"); assert.equal(envelope.data.schema.registered, true); - assert.equal(envelope.data.schema.id, EVAL_PUBLICATION_STATE_SCHEMA_ID); + assert.equal(envelope.data.schema.id, EVAL_GROUND_TRUTH_SCHEMA_ID); assert.equal(envelope.data.schema.bundle_sha256, evalSchemaBundleDigest()); - const renamedSchema = path.join(temporary, "renamed-publication-state.schema.json"); + const renamedSchema = path.join(temporary, "renamed-ground-truth.schema.json"); fs.copyFileSync(schema, renamedSchema); const renamedCapture = await capture([ "json", @@ -700,7 +696,7 @@ test("json validate recognizes the pinned eval schema and reports the owning eva "--schema", renamedSchema, "--file", - publicationState, + groundTruth, "--json" ]); assert.equal(renamedCapture.code, 0); @@ -708,12 +704,12 @@ test("json validate recognizes the pinned eval schema and reports the owning eva data: { schema: { id: string; bundle_sha256: string; registered: boolean } }; }; assert.equal(renamedEnvelope.data.schema.registered, true); - assert.equal(renamedEnvelope.data.schema.id, EVAL_PUBLICATION_STATE_SCHEMA_ID); + assert.equal(renamedEnvelope.data.schema.id, EVAL_GROUND_TRUTH_SCHEMA_ID); assert.equal(renamedEnvelope.data.schema.bundle_sha256, evalSchemaBundleDigest()); - const tamperedSchema = path.join(temporary, "eval-publication-state.schema.json"); + const tamperedSchema = path.join(temporary, "eval-ground-truth.schema.json"); fs.writeFileSync(tamperedSchema, `${fs.readFileSync(schema, "utf8")} `, "utf8"); - const tamperedCapture = await capture(["json", "validate", "--schema", tamperedSchema, "--file", publicationState]); + const tamperedCapture = await capture(["json", "validate", "--schema", tamperedSchema, "--file", groundTruth]); assert.equal(tamperedCapture.code, 2); assert.match(tamperedCapture.stderr, /JSON_SCHEMA_DIGEST_MISMATCH/u); } finally { diff --git a/packages/cli/test/lifecycle-commands.test.ts b/packages/cli/test/lifecycle-commands.test.ts index 66bc223b5..9bb832fa1 100644 --- a/packages/cli/test/lifecycle-commands.test.ts +++ b/packages/cli/test/lifecycle-commands.test.ts @@ -1,10 +1,10 @@ import assert from "node:assert/strict"; import fs from "node:fs"; -import os from "node:os"; import path from "node:path"; -import test from "node:test"; +import test, { type TestContext } from "node:test"; import { runCli } from "../src/index.js"; +import { temporaryRoot } from "./temporary-root.js"; const WORKFLOW_RUN_ID = "ultrafuzz-lifecycle-cli-run"; const RUN_ID = "lifecycle-cli-run"; @@ -15,8 +15,8 @@ interface Capture { code: number; } -function tempProject(): string { - return fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ufz-lifecycle-cli-")); +function tempProject(t: TestContext): string { + return temporaryRoot("ufz-lifecycle-cli-", t); } function shellQuote(value: string): string { @@ -372,9 +372,10 @@ function fakeEnv(project: string, options: { cancelStatus?: string } = {}): Reco } async function launchedProject( + t: TestContext, options: { cancelStatus?: string } = {} ): Promise<{ project: string; env: Record; runRoot: string }> { - const project = tempProject(); + const project = tempProject(t); const env = fakeEnv(project, options); const init = await cli(project, ["init", "--json"], env); assert.equal(init.code, 0, init.stderr); @@ -385,8 +386,8 @@ async function launchedProject( return { project, env, runRoot: runData.run_root }; } -test("why reports the diagnosis in human and JSON output", async () => { - const { project, env } = await launchedProject(); +test("why reports the diagnosis in human and JSON output", async (t) => { + const { project, env } = await launchedProject(t); const human = await cli(project, ["why", RUN_ID], env); assert.equal(human.code, 0, human.stderr); @@ -408,8 +409,8 @@ test("why reports the diagnosis in human and JSON output", async () => { assert.equal(data.blockers[1]?.node_id, "node:strategy"); }); -test("timeline surfaces frame numbers for fork --frame", async () => { - const { project, env } = await launchedProject(); +test("timeline surfaces frame numbers for fork --frame", async (t) => { + const { project, env } = await launchedProject(t); const human = await cli(project, ["timeline", RUN_ID], env); assert.equal(human.code, 0, human.stderr); @@ -429,8 +430,8 @@ test("timeline surfaces frame numbers for fork --frame", async () => { ); }); -test("snapshots lists checkpoints without engine-internal identifiers", async () => { - const { project, env } = await launchedProject(); +test("snapshots lists checkpoints without engine-internal identifiers", async (t) => { + const { project, env } = await launchedProject(t); const human = await cli(project, ["snapshots", RUN_ID], env); assert.equal(human.code, 0, human.stderr); @@ -444,8 +445,8 @@ test("snapshots lists checkpoints without engine-internal identifiers", async () assert.doesNotMatch(JSON.stringify(body), /commit-1|op-1|workspace/u); }); -test("events returns bounded lifecycle events in human and JSON output", async () => { - const { project, env } = await launchedProject(); +test("events returns bounded lifecycle events in human and JSON output", async (t) => { + const { project, env } = await launchedProject(t); const human = await cli(project, ["events", RUN_ID], env); assert.equal(human.code, 0, human.stderr); @@ -460,8 +461,8 @@ test("events returns bounded lifecycle events in human and JSON output", async ( assert.doesNotMatch(fs.readFileSync(path.join(project, "smithers-commands.log"), "utf8"), /--raw/u); }); -test("events --watch streams one line per event and terminates", async () => { - const { project, env } = await launchedProject(); +test("events --watch streams one line per event and terminates", async (t) => { + const { project, env } = await launchedProject(t); const human = await cli(project, ["events", RUN_ID, "--watch", "--interval", "1"], env); assert.equal(human.code, 0, human.stderr); @@ -482,8 +483,8 @@ test("events --watch streams one line per event and terminates", async () => { } }); -test("node reports focused status and only expands tool payloads with --tools", async () => { - const { project, env } = await launchedProject(); +test("node reports focused status and only expands tool payloads with --tools", async (t) => { + const { project, env } = await launchedProject(t); const human = await cli(project, ["node", RUN_ID, "node:project-discovery"], env); assert.equal(human.code, 0, human.stderr); @@ -510,8 +511,8 @@ test("node reports focused status and only expands tool payloads with --tools", assert.deepEqual(data.attempts[0]?.tool_calls[0]?.input, { command: "forge build" }); }); -test("node --watch emits NDJSON envelopes and terminates", async () => { - const { project, env } = await launchedProject(); +test("node --watch emits NDJSON envelopes and terminates", async (t) => { + const { project, env } = await launchedProject(t); const watched = await cli(project, ["node", RUN_ID, "node:project-discovery", "--watch", "--json"], env); @@ -527,8 +528,8 @@ test("node --watch emits NDJSON envelopes and terminates", async () => { ); }); -test("events rejects a raw event category instead of widening the view", async () => { - const { project, env } = await launchedProject(); +test("events rejects a raw event category instead of widening the view", async (t) => { + const { project, env } = await launchedProject(t); const rejected = await cli(project, ["events", RUN_ID, "--type", "agent", "--json"], env); @@ -544,8 +545,8 @@ test("events rejects a raw event category instead of widening the view", async ( assert.equal(accepted.code, 0, accepted.stderr); }); -test("events --watch --json keeps a stream failure on one NDJSON line", async () => { - const { project, env } = await launchedProject(); +test("events --watch --json keeps a stream failure on one NDJSON line", async (t) => { + const { project, env } = await launchedProject(t); fs.writeFileSync(path.join(project, "fake-events-failure"), "stream broke\n", "utf8"); const watched = await cli(project, ["events", RUN_ID, "--watch", "--json"], env); @@ -559,8 +560,8 @@ test("events --watch --json keeps a stream failure on one NDJSON line", async () assertNoEngineBranding(body); }); -test("cancel distinguishes a submitted request from a confirmed cancellation", async () => { - const requested = await launchedProject({ cancelStatus: "cancel-requested" }); +test("cancel distinguishes a submitted request from a confirmed cancellation", async (t) => { + const requested = await launchedProject(t, { cancelStatus: "cancel-requested" }); const human = await cli(requested.project, ["cancel", RUN_ID], requested.env); assert.equal(human.code, 0, human.stderr); @@ -570,7 +571,7 @@ test("cancel distinguishes a submitted request from a confirmed cancellation", a }; assert.equal(runningState.status, "running"); - const confirmed = await launchedProject({ cancelStatus: "cancelled" }); + const confirmed = await launchedProject(t, { cancelStatus: "cancelled" }); const json = await cli(confirmed.project, ["cancel", RUN_ID, "--json"], confirmed.env); assert.equal(json.code, 0, json.stderr); const body = parseJson(json); @@ -584,8 +585,8 @@ test("cancel distinguishes a submitted request from a confirmed cancellation", a assert.match(humanConfirmed.stdout, /^Cancellation confirmed: lifecycle-cli-run is canceled$/mu); }); -test("doctor reports install posture in human and JSON output", async () => { - const { project, env } = await launchedProject(); +test("doctor reports install posture in human and JSON output", async (t) => { + const { project, env } = await launchedProject(t); writeSmallTopology(project, "recon"); const doctorEnv = { ...env, PATH: path.dirname(env.SMITHERS_BIN!) }; @@ -601,7 +602,9 @@ test("doctor reports install posture in human and JSON output", async () => { assert.match(human.stdout + human.stderr, /- compatibility patches: .*supervisor descriptor /u); assert.match(human.stdout + human.stderr, /- compatibility patches: .*terminal state restore /u); assert.match(human.stdout + human.stderr, /- compatibility patches: .*resume hydration /u); - assert.match(human.stdout + human.stderr, /- recon: missing from execution environment/u); + assert.match(human.stdout + human.stderr, /^- recon: missing from execution environment$/mu); + // The scaffold's opt-in Kimi profile is configured but not selected by the topology. + assert.match(human.stdout + human.stderr, /^- kimi: missing from execution environment \(not required\)$/mu); assert.doesNotMatch(human.stdout + human.stderr, /smthrs/u); assert.doesNotMatch(human.stdout + human.stderr, /smthrs/iu); @@ -642,8 +645,8 @@ test("doctor reports install posture in human and JSON output", async () => { assert.ok(overrideBody.diagnostics.some((diagnostic) => diagnostic.code === "DOCTOR_AGENT_CREDENTIAL_MISSING")); }); -test("status recommends ultrafuzz why instead of the engine command", async () => { - const project = tempProject(); +test("status recommends ultrafuzz why instead of the engine command", async (t) => { + const project = tempProject(t); const binDir = path.join(path.dirname(project), path.basename(project) + "-fake-bin"); fs.mkdirSync(binDir, { recursive: true }); const smithers = path.join(binDir, "smithers"); diff --git a/packages/cli/test/temporary-root.test.ts b/packages/cli/test/temporary-root.test.ts new file mode 100644 index 000000000..9cd8ba9f1 --- /dev/null +++ b/packages/cli/test/temporary-root.test.ts @@ -0,0 +1,26 @@ +import assert from "node:assert/strict"; +import fs from "node:fs"; +import path from "node:path"; +import test from "node:test"; + +import { temporaryRoot } from "./temporary-root.js"; + +test("temporaryRoot removes a sealed run tree and the fake-bin sibling when the test ends", async (t) => { + let root = ""; + await t.test("fixture", (inner) => { + root = temporaryRoot("ufz-cli-cleanup-", inner); + // The shape a launched run leaves: read-only files in dr-x directories. + const sealed = path.join(root, ".ultrafuzz", "runs", "run-1", "snapshot"); + fs.mkdirSync(sealed, { recursive: true }); + fs.writeFileSync(path.join(sealed, "module.js"), "export {};\n", { mode: 0o400 }); + for (let directory = sealed; directory !== root; directory = path.dirname(directory)) { + fs.chmodSync(directory, 0o500); + } + fs.mkdirSync(`${root}-fake-bin`); + fs.writeFileSync(path.join(`${root}-fake-bin`, "smithers"), "#!/bin/sh\n", { mode: 0o500 }); + }); + + assert.notEqual(root, ""); + assert.equal(fs.existsSync(root), false); + assert.equal(fs.existsSync(`${root}-fake-bin`), false); +}); diff --git a/packages/cli/test/temporary-root.ts b/packages/cli/test/temporary-root.ts new file mode 100644 index 000000000..cca32c14e --- /dev/null +++ b/packages/cli/test/temporary-root.ts @@ -0,0 +1,31 @@ +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import type { TestContext } from "node:test"; + +/** + * Create a canonical temporary directory that is removed when test `t` ends, + * together with the `-fake-bin` directory the fake engine fixtures create + * beside it. A launched run leaves about 450 MB of sealed snapshot directories + * with mode `dr-x`, which `rmSync` cannot remove until owner write permission is + * restored on the way down. + */ +export function temporaryRoot(prefix: string, t: TestContext): string { + const root = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), prefix)); + t.after(() => { + for (const directory of [root, `${root}-fake-bin`]) { + restoreOwnerWrite(directory); + fs.rmSync(directory, { recursive: true, force: true, maxRetries: 3 }); + } + }); + return root; +} + +function restoreOwnerWrite(directory: string): void { + const stat = fs.lstatSync(directory, { throwIfNoEntry: false }); + if (stat === undefined || !stat.isDirectory()) return; + fs.chmodSync(directory, stat.mode | 0o700); + for (const entry of fs.readdirSync(directory, { withFileTypes: true })) { + if (entry.isDirectory()) restoreOwnerWrite(path.join(directory, entry.name)); + } +} diff --git a/packages/config/package.json b/packages/config/package.json index 4297e5bdf..ab5b6866d 100644 --- a/packages/config/package.json +++ b/packages/config/package.json @@ -18,7 +18,7 @@ ], "scripts": { "build": "tsc -p tsconfig.build.json && cp defaults.toml dist/ultrafuzz.toml && cp audit-profiles.yml dist/audit-profiles.yml && rm -rf dist/topologies && cp -R topologies dist/topologies && rm -rf dist/schema && cp -R schema dist/schema && node scripts/verify-schema-registry.mjs", - "test": "pnpm --filter @ultrafuzz/config^... build && vitest run", + "test": "pnpm --filter @ultrafuzz/config^... build && vitest run --testTimeout=30000", "typecheck": "pnpm --filter @ultrafuzz/config^... build && tsc -p tsconfig.json --noEmit --pretty false" }, "dependencies": { diff --git a/packages/config/src/config-schema-registry.ts b/packages/config/src/config-schema-registry.ts index 1683a58bc..3dfd63ef1 100644 --- a/packages/config/src/config-schema-registry.ts +++ b/packages/config/src/config-schema-registry.ts @@ -15,11 +15,7 @@ import { type SchemaRegistryEntry } from "@ultrafuzz/artifacts"; -import { - RESOLVED_CONFIG_JSON_SCHEMA_ID, - RESOLVED_CONFIG_SCHEMA_FILENAME, - resolvedConfigZodSchema -} from "./resolved-config-schema.js"; +import { RESOLVED_CONFIG_JSON_SCHEMA_ID, RESOLVED_CONFIG_SCHEMA_FILENAME } from "./resolved-config-schema.js"; import type { ResolvedConfig } from "./types.js"; import { isRecord } from "@ultrafuzz/artifacts"; @@ -142,13 +138,6 @@ export function configSchemaBundleDigest(): string { return schemaRegistryBundleDigest(configSchemaRegistry()); } -export function resolvedConfigSchemaEntry(): SchemaRegistryEntry { - const entry = configSchemaRegistry().find((candidate) => candidate.id === RESOLVED_CONFIG_JSON_SCHEMA_ID); - if (entry === undefined) - throw new Error(`registered config schema is unavailable: ${RESOLVED_CONFIG_JSON_SCHEMA_ID}`); - return entry; -} - export function validateResolvedConfigJson(value: unknown): JsonSchemaValidationResult { const validator = configValidator().getSchema(RESOLVED_CONFIG_JSON_SCHEMA_ID); if (validator === undefined) @@ -196,10 +185,6 @@ export function serializeResolvedConfigJsonBytes(config: ResolvedConfig): Buffer return bytes; } -export function resolvedConfigValidatorsAgree(value: unknown): boolean { - return validateResolvedConfigJson(value).ok === resolvedConfigZodSchema.safeParse(value).success; -} - function configValidator(): ReturnType { if (cachedValidator !== undefined) return cachedValidator; const registry = configSchemaRegistry(); diff --git a/packages/config/src/defaults.ts b/packages/config/src/defaults.ts index 62d6e3285..cab776c99 100644 --- a/packages/config/src/defaults.ts +++ b/packages/config/src/defaults.ts @@ -15,7 +15,6 @@ import { type ModelProfile, type PermissionConfig, type ProjectConfigInput, - type PromptMetadataLayer, type RetryConfig, type ResolvedConfig, type RunConfig @@ -37,10 +36,6 @@ export function createDefaultResolvedConfig(): ResolvedConfig { return cloneResolvedConfig(DEFAULT_CONFIG); } -export function createDefaultPromptMetadataLayer(): PromptMetadataLayer { - return {}; -} - export function synthesizeDefaultModelProfile(agent = DEFAULT_AGENT): ModelProfile { return { id: DEFAULT_MODEL_PROFILE_ID, diff --git a/packages/config/src/loader.ts b/packages/config/src/loader.ts index 3deeb87b7..cfdb4af0e 100644 --- a/packages/config/src/loader.ts +++ b/packages/config/src/loader.ts @@ -121,7 +121,10 @@ export async function loadProjectConfig( export function parseProjectConfigToml(text: string, file = CONFIG_FILE_NAME): ConfigResult { let table: TomlTable; try { - table = parse(text); + // smol-toml 1.9 builds every table with Object.create(null), which this + // loader's plain-object checks and entry sorts do not accept. structuredClone + // returns the same data as ordinary objects. + table = structuredClone(parse(text)); } catch (error) { return fail([ diagnostic("CONFIG_TOML_PARSE_FAILED", `${file} is not valid TOML`, [], "project-toml", tomlLocation(error, file)) diff --git a/packages/config/src/model-profiles.ts b/packages/config/src/model-profiles.ts index 26d9f7ca4..3bffb3a01 100644 --- a/packages/config/src/model-profiles.ts +++ b/packages/config/src/model-profiles.ts @@ -66,10 +66,6 @@ export function validProfileId(id: string): boolean { return profileIdSchema.safeParse(id).success; } -export function applyDefaultProfileOverrides(config: ResolvedConfig, overrides: DefaultProfileOverrides): void { - applyModelProfileOverrides(config, config.models.default, overrides); -} - export function applyModelProfileOverrides( config: ResolvedConfig, profileId: string, diff --git a/packages/config/src/redaction.ts b/packages/config/src/redaction.ts index b382f670c..462cd4505 100644 --- a/packages/config/src/redaction.ts +++ b/packages/config/src/redaction.ts @@ -1,20 +1,7 @@ import { cloneResolvedConfig } from "./defaults.js"; import { serializeResolvedConfigToml, type SerializeResolvedConfigTomlOptions } from "./resolve.js"; -import { - SENSITIVE_REDACTION_PLACEHOLDER, - hasRedactionPlaceholder, - isSensitiveSecretValue, - redactSecretsInText -} from "@ultrafuzz/security"; -import { - diagnostic, - fail, - hasErrors, - ok, - type ConfigDiagnostic, - type ConfigResult, - type ResolvedConfig -} from "./types.js"; +import { SENSITIVE_REDACTION_PLACEHOLDER, isSensitiveSecretValue, redactSecretsInText } from "@ultrafuzz/security"; +import type { ConfigDiagnostic, ResolvedConfig } from "./types.js"; export const REDACTION_PLACEHOLDER = SENSITIVE_REDACTION_PLACEHOLDER; export const CONFIG_REDACTIONS_SCHEMA_VERSION = "ultrafuzz.config-redactions.v2" as const; @@ -64,54 +51,6 @@ export function serializeRedactedResolvedConfigToml( return serializeResolvedConfigToml("config" in redacted ? redacted.config : redacted, options); } -export function restoreRedactedConfig( - redactedConfig: ResolvedConfig, - currentConfig: ResolvedConfig, - manifest: RedactionManifest -): ConfigResult { - const restored = cloneResolvedConfig(redactedConfig); - const diagnostics: ConfigDiagnostic[] = []; - - for (const entry of manifest.entries) { - const currentValue = getPath(currentConfig, entry.path); - if (typeof currentValue !== "string" || currentValue.length === 0 || hasRedactionPlaceholder(currentValue)) { - diagnostics.push( - diagnostic( - "CONFIG_REDACTION_RESTORE_MISSING", - `redacted value at ${entry.key} must be restored from current config before workflow launch`, - entry.path, - "redaction" - ) - ); - continue; - } - setPath(restored, entry.path, currentValue); - } - - diagnostics.push(...assertNoRedactionPlaceholders(restored).diagnostics); - if (hasErrors(diagnostics)) { - return fail(diagnostics); - } - return ok(restored); -} - -export function assertNoRedactionPlaceholders(config: ResolvedConfig): ConfigResult { - const diagnostics: ConfigDiagnostic[] = []; - visitStrings(config, [], (path, value) => { - if (hasRedactionPlaceholder(value)) { - diagnostics.push( - diagnostic( - "CONFIG_REDACTION_PLACEHOLDER_PRESENT", - `redaction placeholder at ${path.join(".")} cannot be passed to workflow launch`, - path, - "redaction" - ) - ); - } - }); - return diagnostics.length > 0 ? fail(diagnostics) : ok(undefined); -} - export function redactDiagnostics(diagnostics: ConfigDiagnostic[]): ConfigDiagnostic[] { return diagnostics.map((entry) => ({ ...entry, @@ -139,59 +78,6 @@ function redactSensitiveScalar( } } -function visitStrings(value: unknown, path: string[], visitor: (path: string[], value: string) => void): void { - if (typeof value === "string") { - visitor(path, value); - return; - } - if (Array.isArray(value)) { - value.forEach((entry, index) => { - visitStrings(entry, path.concat(String(index)), visitor); - }); - return; - } - if (value && typeof value === "object") { - for (const [key, child] of Object.entries(value)) { - visitStrings(child, path.concat(key), visitor); - } - } -} - -function getPath(value: unknown, path: string[]): unknown { - let current = value; - for (const segment of path) { - if (Array.isArray(current)) { - current = current[Number(segment)]; - } else if (current && typeof current === "object") { - current = (current as Record)[segment]; - } else { - return undefined; - } - } - return current; -} - -function setPath(value: unknown, path: string[], nextValue: string): void { - let current = value as Record; - for (const segment of path.slice(0, -1)) { - const child = current[segment]; - if (Array.isArray(child)) { - current = child as unknown as Record; - } else { - current = child as Record; - } - } - const last = path[path.length - 1]; - if (last === undefined) { - return; - } - if (Array.isArray(current)) { - current[Number(last)] = nextValue; - } else { - current[last] = nextValue; - } -} - function formatPath(path: string[]): string { return path.join("."); } diff --git a/packages/config/src/resolve.ts b/packages/config/src/resolve.ts index c431a1ced..797c33fd5 100644 --- a/packages/config/src/resolve.ts +++ b/packages/config/src/resolve.ts @@ -5,7 +5,6 @@ import { DEFAULT_MODEL_PROFILE_ID, MAX_TIMEOUT_SECONDS, cloneResolvedConfig, - createDefaultPromptMetadataLayer, createDefaultResolvedConfig } from "./defaults.js"; import { @@ -33,7 +32,6 @@ import { type ConfigResult, type PermissionConfig, type ProjectConfigInput, - type PromptMetadataLayer, type ResolveConfigInput, type ResolvedConfig, type RuntimeConfigOverrides @@ -71,10 +69,6 @@ export function resolveConfig(input: ResolveConfigInput = {}): ConfigResult): void { if (layer.mode !== undefined) { config.execution.mode = layer.mode; diff --git a/packages/config/src/resolved-config-schema.ts b/packages/config/src/resolved-config-schema.ts index e7d1e095b..76cd63583 100644 --- a/packages/config/src/resolved-config-schema.ts +++ b/packages/config/src/resolved-config-schema.ts @@ -365,16 +365,3 @@ export const resolvedConfigZodSchema: z.ZodType = z }); } }); - -export function assertResolvedConfigZod( - value: unknown, - label = "resolved configuration" -): asserts value is ResolvedConfig { - const result = resolvedConfigZodSchema.safeParse(value); - if (result.success) return; - const summary = result.error.issues - .slice(0, 10) - .map((issue) => `${issue.path.join(".") || "/"}: ${issue.message}`) - .join("; "); - throw new Error(`${label} does not match ${RESOLVED_CONFIG_JSON_SCHEMA_ID}: ${summary}`); -} diff --git a/packages/config/src/types.ts b/packages/config/src/types.ts index 772f9c6f4..511339ab0 100644 --- a/packages/config/src/types.ts +++ b/packages/config/src/types.ts @@ -3,14 +3,7 @@ import type { AuditProfileSettings, DynamicStrategiesEnumerator } from "./audit- export type DiagnosticSeverity = "error" | "warning"; export type ConfigDiagnosticSource = - | "defaults" - | "audit-profile" - | "prompt-metadata" - | "project-toml" - | "environment" - | "runtime" - | "validation" - | "redaction"; + "defaults" | "audit-profile" | "project-toml" | "environment" | "runtime" | "validation"; export interface ConfigDiagnostic { code: string; @@ -240,11 +233,6 @@ export interface AuditProfileResolution { export type AuditProfileSettingOrigin = "default" | "audit-profile" | "project-config" | "environment" | "runtime-override"; -export interface PromptMetadataLayer { - models?: Record & { id?: string }>; - run?: Partial; -} - export interface ProjectConfigInput { schemaVersion?: typeof PROJECT_CONFIG_SCHEMA_VERSION; auditProfile?: string; @@ -304,7 +292,6 @@ export interface RuntimeConfigOverrides extends ProjectConfigInput { export interface ResolveConfigInput { projectConfig?: ProjectConfigInput; - promptMetadata?: PromptMetadataLayer; env?: Record; runtimeOverrides?: RuntimeConfigOverrides; } diff --git a/packages/config/test/config.test.ts b/packages/config/test/config.test.ts index 2c1a976d5..5fd260040 100644 --- a/packages/config/test/config.test.ts +++ b/packages/config/test/config.test.ts @@ -11,8 +11,7 @@ import { MODAL_NODE_MAX_INNER_TIMEOUT_SECONDS, MODAL_SANDBOX_MAX_LIFETIME_SECONDS, REDACTION_PLACEHOLDER, - assertNoRedactionPlaceholders, - applyDefaultProfileOverrides, + applyModelProfileOverrides, invariantPropertyPrioritySelection, loadProjectConfig, parseProjectConfigToml, @@ -20,7 +19,6 @@ import { redactResolvedConfig, resolveExecutionResources, resolveConfig, - restoreRedactedConfig, serializeRedactedResolvedConfigToml, type ConfigDiagnostic, type ProjectConfigInput, @@ -547,7 +545,7 @@ credential_env = ["MODAL_TOKEN_ID", "MODAL_TOKEN_SECRET"] } }); - it("applies defaults, prompt metadata, project TOML, env, then runtime overrides", () => { + it("applies defaults, project TOML, env, then runtime overrides", () => { const project = parseProjectConfigToml(` schema_version = "ultrafuzz.config.v2" dynamic_strategies_enumerator = 5 @@ -578,15 +576,6 @@ config_dir = "teams/codex" if (!project.ok) return; const resolved = resolveConfig({ - promptMetadata: { - run: { defaultTimeoutSeconds: 900 }, - models: { - "prompt-model": { - agent: "CodexAgent", - model: "gpt-5.5" - } - } - }, projectConfig: project.value, env: { ULTRAFUZZ_MAX_PARALLEL_AGENTS: "7", @@ -820,7 +809,7 @@ output_dir = ".ultrafuzz/custom-runs" }); describe("redaction", () => { - it("redacts sensitive model values while preserving restore requirements", () => { + it("redacts sensitive model values from the persisted config and records them in the manifest", () => { const resolved = resolveConfig({ env: {}, projectConfig: { @@ -845,50 +834,7 @@ describe("redaction", () => { expect(toml).not.toContain("sk-test-secret"); expect(redacted.manifest.entries.map((entry) => entry.key)).toContain("models.profiles.secret-model.model"); expect(redacted.manifest.entries[0]?.requiredForWorkflowLaunch).toBe(true); - - const restored = restoreRedactedConfig(redacted.config, resolved.value, redacted.manifest); - expect(restored.ok).toBe(true); - if (!restored.ok) return; - expect(restored.value.models.profiles["secret-model"]?.model).toBe("sk-test-secret"); - expect(assertNoRedactionPlaceholders(restored.value).ok).toBe(true); - }); - - it("rejects literal redaction placeholders before workflow launch", () => { - const resolved = resolveConfig({ env: {} }); - expect(resolved.ok).toBe(true); - if (!resolved.ok) return; - resolved.value.models.profiles.default!.model = "[redacted]"; - const checked = assertNoRedactionPlaceholders(resolved.value); - expect(checked.ok).toBe(false); - expect(checked.diagnostics[0]?.code).toBe("CONFIG_REDACTION_PLACEHOLDER_PRESENT"); - }); - - it("fails restoration when a required redacted value is unavailable", () => { - const resolved = resolveConfig({ - env: {}, - projectConfig: { - models: { - profiles: { - "secret-model": { - agent: "CodexAgent", - model: "sk-test-secret" - } - }, - default: "secret-model" - } - } - }); - expect(resolved.ok).toBe(true); - if (!resolved.ok) throw new Error(JSON.stringify(resolved.diagnostics, null, 2)); - - const redacted = redactResolvedConfig(resolved.value); - const current = structuredClone(resolved.value); - delete current.models.profiles["secret-model"]!.model; - - const restored = restoreRedactedConfig(redacted.config, current, redacted.manifest); - expect(restored.ok).toBe(false); - expect(restored.diagnostics.map((entry) => entry.code)).toContain("CONFIG_REDACTION_RESTORE_MISSING"); - expect(restored.diagnostics[0]?.message).toContain("before workflow launch"); + expect(resolved.value.models.profiles["secret-model"]?.model).toBe("sk-test-secret"); }); }); @@ -1054,7 +1000,7 @@ describe("model profile and triage validation", () => { const agentOnly = resolveConfig({ env: {} }); expect(agentOnly.ok).toBe(true); if (!agentOnly.ok) return; - applyDefaultProfileOverrides(agentOnly.value, { agent: "ClaudeAgent" }); + applyModelProfileOverrides(agentOnly.value, agentOnly.value.models.default, { agent: "ClaudeAgent" }); expect(agentOnly.value.models.profiles.default).toMatchObject({ agent: "ClaudeAgent" }); @@ -1064,7 +1010,10 @@ describe("model profile and triage validation", () => { const pinned = resolveConfig({ env: {} }); expect(pinned.ok).toBe(true); if (!pinned.ok) return; - applyDefaultProfileOverrides(pinned.value, { agent: "ClaudeAgent", model: "claude-sonnet-5" }); + applyModelProfileOverrides(pinned.value, pinned.value.models.default, { + agent: "ClaudeAgent", + model: "claude-sonnet-5" + }); expect(pinned.value.models.profiles.default).toMatchObject({ agent: "ClaudeAgent", model: "claude-sonnet-5" @@ -1074,7 +1023,7 @@ describe("model profile and triage validation", () => { const benchmark = resolveConfig({ env: {} }); expect(benchmark.ok).toBe(true); if (!benchmark.ok) return; - applyDefaultProfileOverrides(benchmark.value, { + applyModelProfileOverrides(benchmark.value, benchmark.value.models.default, { agent: "CodexAgent", model: "gpt-5.6-luna", reasoning: "high" diff --git a/packages/config/test/resolved-config-schema.test.ts b/packages/config/test/resolved-config-schema.test.ts index f83d529cd..51a0544bd 100644 --- a/packages/config/test/resolved-config-schema.test.ts +++ b/packages/config/test/resolved-config-schema.test.ts @@ -17,7 +17,6 @@ import { parseProjectConfigToml, parseResolvedConfigJsonBytes, resolvedConfigJsonSchema, - resolvedConfigValidatorsAgree, resolvedConfigZodSchema, resolveConfig, serializeResolvedConfigJsonBytes, @@ -320,6 +319,10 @@ describe("resolved config JSON contract", () => { ); }); +function resolvedConfigValidatorsAgree(value: unknown): boolean { + return validateResolvedConfigJson(value).ok === resolvedConfigZodSchema.safeParse(value).success; +} + function fixturePath(filename: string): string { return path.join(path.dirname(fileURLToPath(import.meta.url)), "fixtures", filename); } diff --git a/packages/config/test/toml-null-prototype.test.ts b/packages/config/test/toml-null-prototype.test.ts new file mode 100644 index 000000000..0d9e544f8 --- /dev/null +++ b/packages/config/test/toml-null-prototype.test.ts @@ -0,0 +1,42 @@ +import fs from "node:fs"; + +import type * as SmolToml from "smol-toml"; +import { describe, expect, it, vi } from "vitest"; + +// smol-toml 1.9 builds every table with Object.create(null). The workspace +// lockfile resolves 1.7, but a packed install resolves the newest 1.x, so +// reproduce the 1.9 shape here. +vi.mock("smol-toml", async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, parse: (text: string) => withNullPrototypes(actual.parse(text)) }; +}); + +import { createDefaultResolvedConfig, parseProjectConfigToml } from "../src/index.js"; + +function withNullPrototypes(value: unknown): unknown { + if (Array.isArray(value)) return value.map(withNullPrototypes); + if (typeof value !== "object" || value === null || value instanceof Date) return value; + const table = Object.create(null) as Record; + for (const [key, entry] of Object.entries(value)) table[key] = withNullPrototypes(entry); + return table; +} + +describe("TOML tables without a prototype", () => { + it("load the shipped defaults and a project config", () => { + const defaults = parseProjectConfigToml(fs.readFileSync(new URL("../defaults.toml", import.meta.url), "utf8")); + expect(defaults.ok, JSON.stringify(defaults.diagnostics)).toBe(true); + expect(createDefaultResolvedConfig().models.default).toBe("default"); + + const project = parseProjectConfigToml(` +[models.fast] +agent = "CodexAgent" +model = "gpt-5.5" + +[agents.CodexAgent] +auth = "subscription" +`); + expect(project.ok, JSON.stringify(project.diagnostics)).toBe(true); + if (!project.ok) return; + expect(project.value.models?.profiles?.fast).toMatchObject({ agent: "CodexAgent", model: "gpt-5.5" }); + }); +}); diff --git a/packages/config/topologies/default.yml b/packages/config/topologies/default.yml index 3e627b41d..c763c203a 100644 --- a/packages/config/topologies/default.yml +++ b/packages/config/topologies/default.yml @@ -41,6 +41,8 @@ groups: review: label: Review color: "#0f766e" + defaults: + timeout_seconds: 7200 nodes: - id: __start__ kind: meta @@ -198,7 +200,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/0kn0t.md contract: ultrafuzz/nonempty-markdown@1 @@ -211,7 +212,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/certora-thinking.md contract: ultrafuzz/nonempty-markdown@1 @@ -224,7 +224,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/certora-sanity.md contract: ultrafuzz/nonempty-markdown@1 @@ -237,7 +236,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/aviggiano.md contract: ultrafuzz/nonempty-markdown@1 @@ -250,7 +248,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/rounding.md contract: ultrafuzz/nonempty-markdown@1 @@ -263,7 +260,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/crytic.md contract: ultrafuzz/nonempty-markdown@1 @@ -276,7 +272,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/runtime-verification.md contract: ultrafuzz/nonempty-markdown@1 @@ -289,7 +284,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/a16z-erc4626.md contract: ultrafuzz/nonempty-markdown@1 @@ -302,7 +296,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/recon.md contract: ultrafuzz/nonempty-markdown@1 diff --git a/packages/config/topologies/exhaustive.yml b/packages/config/topologies/exhaustive.yml index ac088eaff..3dafaa7ec 100644 --- a/packages/config/topologies/exhaustive.yml +++ b/packages/config/topologies/exhaustive.yml @@ -41,6 +41,8 @@ groups: review: label: Review color: "#0f766e" + defaults: + timeout_seconds: 7200 nodes: - id: __start__ kind: meta @@ -198,7 +200,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/0kn0t.md contract: ultrafuzz/nonempty-markdown@1 @@ -211,7 +212,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/certora-thinking.md contract: ultrafuzz/nonempty-markdown@1 @@ -224,7 +224,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/certora-sanity.md contract: ultrafuzz/nonempty-markdown@1 @@ -237,7 +236,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/aviggiano.md contract: ultrafuzz/nonempty-markdown@1 @@ -250,7 +248,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/rounding.md contract: ultrafuzz/nonempty-markdown@1 @@ -263,7 +260,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/crytic.md contract: ultrafuzz/nonempty-markdown@1 @@ -276,7 +272,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/runtime-verification.md contract: ultrafuzz/nonempty-markdown@1 @@ -289,7 +284,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/a16z-erc4626.md contract: ultrafuzz/nonempty-markdown@1 @@ -302,7 +296,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/recon.md contract: ultrafuzz/nonempty-markdown@1 diff --git a/packages/config/topologies/invariant-only.yml b/packages/config/topologies/invariant-only.yml index 90c069448..efc7b0d7f 100644 --- a/packages/config/topologies/invariant-only.yml +++ b/packages/config/topologies/invariant-only.yml @@ -23,6 +23,8 @@ groups: review: label: Review color: "#0f766e" + defaults: + timeout_seconds: 7200 nodes: - id: __start__ kind: meta @@ -89,7 +91,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/0kn0t.md contract: ultrafuzz/nonempty-markdown@1 @@ -102,7 +103,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/certora-thinking.md contract: ultrafuzz/nonempty-markdown@1 @@ -115,7 +115,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/certora-sanity.md contract: ultrafuzz/nonempty-markdown@1 @@ -128,7 +127,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/aviggiano.md contract: ultrafuzz/nonempty-markdown@1 @@ -141,7 +139,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/rounding.md contract: ultrafuzz/nonempty-markdown@1 @@ -154,7 +151,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/crytic.md contract: ultrafuzz/nonempty-markdown@1 @@ -167,7 +163,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/runtime-verification.md contract: ultrafuzz/nonempty-markdown@1 @@ -180,7 +175,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/a16z-erc4626.md contract: ultrafuzz/nonempty-markdown@1 @@ -193,7 +187,6 @@ nodes: group: references depends_on: - __start__ - timeout_seconds: 300 outputs: - path: references/recon.md contract: ultrafuzz/nonempty-markdown@1 diff --git a/packages/dashboard/frontend/src/main.tsx b/packages/dashboard/frontend/src/main.tsx index b3894a5f9..d85821538 100644 --- a/packages/dashboard/frontend/src/main.tsx +++ b/packages/dashboard/frontend/src/main.tsx @@ -223,18 +223,11 @@ type CommandCapabilities = { referencesStatus: boolean; referencesSync: boolean; referencesUpdate: boolean; - restartWholeRun: boolean; status: boolean; - doctor: boolean; config: boolean; report: boolean; - triage: boolean; - merge: boolean; materialize: boolean; clean: boolean; - restartFromNode: boolean; - rerunSelectedNode: boolean; - arbitraryShell: boolean; }; type DashboardFlowNode = Node; diff --git a/packages/dashboard/schema/dashboard-http.schema.json b/packages/dashboard/schema/dashboard-http.schema.json index a40ddfebe..a1e244889 100644 --- a/packages/dashboard/schema/dashboard-http.schema.json +++ b/packages/dashboard/schema/dashboard-http.schema.json @@ -430,14 +430,7 @@ "materialize", "clean", "config", - "restartWholeRun", - "status", - "doctor", - "triage", - "merge", - "restartFromNode", - "rerunSelectedNode", - "arbitraryShell" + "status" ], "properties": { "validate": { "type": "boolean" }, @@ -455,14 +448,7 @@ "materialize": { "type": "boolean" }, "clean": { "type": "boolean" }, "config": { "type": "boolean" }, - "restartWholeRun": { "type": "boolean" }, - "status": { "type": "boolean" }, - "doctor": { "type": "boolean" }, - "triage": { "type": "boolean" }, - "merge": { "type": "boolean" }, - "restartFromNode": { "type": "boolean" }, - "rerunSelectedNode": { "type": "boolean" }, - "arbitraryShell": { "type": "boolean" } + "status": { "type": "boolean" } } }, "findingItem": { diff --git a/packages/dashboard/src/index.ts b/packages/dashboard/src/index.ts index bd68cbbb2..2adc81997 100644 --- a/packages/dashboard/src/index.ts +++ b/packages/dashboard/src/index.ts @@ -1595,14 +1595,7 @@ class DashboardApp { materialize: hasRun, clean: hasRun, config: true, - restartWholeRun: false, - status: hasRun, - doctor: false, - triage: false, - merge: false, - restartFromNode: false, - rerunSelectedNode: false, - arbitraryShell: false + status: hasRun }; } diff --git a/packages/dashboard/test/dashboard.test.ts b/packages/dashboard/test/dashboard.test.ts index 1769793cd..0b4d2f1f8 100644 --- a/packages/dashboard/test/dashboard.test.ts +++ b/packages/dashboard/test/dashboard.test.ts @@ -81,8 +81,6 @@ test("serves logical topology flow with expanded attempt details", async () => { assert.ok(flow.nodes.every((node) => node.id === node.data.logicalNodeId)); assert.equal(flow.capabilities.runNewCampaign, true); assert.equal(flow.capabilities.referencesStatus, true); - assert.equal(flow.capabilities.doctor, false); - assert.equal(flow.capabilities.merge, false); } finally { await handle.close(); } diff --git a/packages/evals/package.json b/packages/evals/package.json index 00074375f..770e0adda 100644 --- a/packages/evals/package.json +++ b/packages/evals/package.json @@ -19,18 +19,14 @@ "scripts": { "build": "rm -rf dist && tsc -p tsconfig.build.json && node scripts/copy-prompts.mjs && rm -rf dist/schema && cp -R schema dist/schema && node scripts/verify-schema-registry.mjs", "schema:check": "pnpm build", - "test": "pnpm --filter @ultrafuzz/evals... build && vitest run", + "test": "pnpm --filter @ultrafuzz/evals... build && vitest run --testTimeout=30000", "typecheck": "pnpm --filter @ultrafuzz/evals^... build && tsc -p tsconfig.json --noEmit --pretty false" }, "dependencies": { "@ultrafuzz/artifacts": "workspace:*", "@ultrafuzz/config": "workspace:*", "@ultrafuzz/runtime": "workspace:*", - "proper-lockfile": "4.1.2", "yaml": "^2.8.0", "zod": "^4.4.3" - }, - "devDependencies": { - "@types/proper-lockfile": "^4.1.4" } } diff --git a/packages/evals/schema/eval-common.schema.json b/packages/evals/schema/eval-common.schema.json index 71c975e1c..b08d0af15 100644 --- a/packages/evals/schema/eval-common.schema.json +++ b/packages/evals/schema/eval-common.schema.json @@ -63,11 +63,6 @@ "propertyNames": { "minLength": 1 }, "additionalProperties": { "type": "string" } }, - "nonNegativeIntegerMap": { - "type": "object", - "propertyNames": { "minLength": 1 }, - "additionalProperties": { "$ref": "#/$defs/nonNegativeInteger" } - }, "runtimeDiagnostic": { "type": "object", "additionalProperties": false, @@ -1340,38 +1335,6 @@ "reviewer_status": { "enum": ["pending", "accepted", "rejected", "needs-more-evidence"] } } }, - "evalPublicationDiagnostic": { - "type": "object", - "additionalProperties": false, - "required": ["code", "row_id", "contract", "reason"], - "properties": { - "code": { "enum": ["TERMINAL_REPORT_NOT_PUBLISHABLE", "RECOVERY_EQUIVALENCE_NOT_PUBLISHABLE"] }, - "row_id": { "$ref": "#/$defs/nonEmptyString" }, - "contract": { "const": "ultrafuzz/report@3" }, - "reason": { "$ref": "#/$defs/nonEmptyString" }, - "report_path": { "$ref": "#/$defs/nonEmptyString" } - } - }, - "evalPublicationState": { - "type": "object", - "additionalProperties": false, - "required": ["schema_version", "status", "diagnostics"], - "properties": { - "schema_version": { "const": "ultrafuzz.eval.publication.v1" }, - "status": { "enum": ["publishable", "non-publishable"] }, - "diagnostics": { - "type": "array", - "items": { "$ref": "#/$defs/evalPublicationDiagnostic" } - } - }, - "allOf": [ - { - "if": { "properties": { "status": { "const": "publishable" } }, "required": ["status"] }, - "then": { "properties": { "diagnostics": { "type": "array", "maxItems": 0 } } }, - "else": { "properties": { "diagnostics": { "type": "array", "minItems": 1 } } } - } - ] - }, "efficiencyCompleteness": { "oneOf": [ { @@ -1617,45 +1580,6 @@ "recovery_equivalence": { "$ref": "#/$defs/recoveryEquivalenceSummary" }, "provenance": { "$ref": "#/$defs/summaryProvenance" } } - }, - "telemetryCursor": { - "type": "object", - "additionalProperties": false, - "required": [ - "schemaVersion", - "byteOffset", - "deliveredEventIds", - "uploadedArtifacts", - "lastHeartbeatAt", - "providerIds", - "findingsCountByNode" - ], - "properties": { - "schemaVersion": { "const": "ultrafuzz.eval.telemetry-cursor.v1" }, - "byteOffset": { "$ref": "#/$defs/nonNegativeInteger" }, - "deliveredEventIds": { - "type": "array", - "items": { "$ref": "#/$defs/nonEmptyString" }, - "maxItems": 4096, - "uniqueItems": true - }, - "uploadedArtifacts": { - "type": "object", - "propertyNames": { "minLength": 1 }, - "additionalProperties": { "type": "string", "pattern": "^[0-9a-f]{64}$" } - }, - "lastHeartbeatAt": { - "type": "object", - "propertyNames": { "minLength": 1 }, - "additionalProperties": { "$ref": "#/$defs/timestamp" } - }, - "providerIds": { - "type": "object", - "propertyNames": { "minLength": 1 }, - "additionalProperties": { "$ref": "#/$defs/nonEmptyString" } - }, - "findingsCountByNode": { "$ref": "#/$defs/nonNegativeIntegerMap" } - } } } } diff --git a/packages/evals/schema/eval-history-automatic-publication-plan.schema.json b/packages/evals/schema/eval-history-automatic-publication-plan.schema.json deleted file mode 100644 index 34048f49b..000000000 --- a/packages/evals/schema/eval-history-automatic-publication-plan.schema.json +++ /dev/null @@ -1,122 +0,0 @@ -{ - "$schema": "https://json-schema.org/draft/2020-12/schema", - "$id": "urn:ultrafuzz:schema:evals:history-automatic-publication-plan:1", - "title": "Ultrafuzz automatic eval-history publication plan v1", - "type": "object", - "additionalProperties": false, - "required": [ - "schema_version", - "candidate_commit", - "candidate_repository_url", - "source_artifact", - "producer_run_id", - "producer_run_attempt", - "mode", - "benchmark", - "pairs" - ], - "properties": { - "schema_version": { "const": "ultrafuzz.eval-history-automatic-publication-plan.v1" }, - "candidate_commit": { "$ref": "#/$defs/fullCommit" }, - "candidate_repository_url": { "$ref": "#/$defs/githubRepository" }, - "source_artifact": { "$ref": "#/$defs/actionsRun" }, - "producer_run_id": { "$ref": "#/$defs/positiveDecimal" }, - "producer_run_attempt": { "$ref": "#/$defs/positiveDecimal" }, - "mode": { "$ref": "#/$defs/lane" }, - "benchmark": { "$ref": "#/$defs/benchmark" }, - "pairs": { - "type": "array", - "minItems": 1, - "maxItems": 4, - "items": { "$ref": "#/$defs/pair" } - } - }, - "$defs": { - "safeId": { - "type": "string", - "pattern": "^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$" - }, - "safeLowerId": { - "type": "string", - "pattern": "^[a-z0-9][a-z0-9._-]{0,127}$" - }, - "fullCommit": { - "type": "string", - "pattern": "^[0-9a-f]{40}$" - }, - "positiveDecimal": { - "type": "string", - "maxLength": 32, - "pattern": "^[1-9][0-9]*$" - }, - "positiveInteger": { - "type": "integer", - "minimum": 1, - "maximum": 9007199254740991 - }, - "githubRepository": { - "type": "string", - "maxLength": 2048, - "pattern": "^https://github\\.com/[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$" - }, - "actionsRun": { - "type": "string", - "maxLength": 2048, - "pattern": "^https://github\\.com/[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+/actions/runs/[1-9][0-9]*$" - }, - "actionsArtifacts": { - "type": "string", - "maxLength": 2048, - "pattern": "^https://github\\.com/[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+/actions/runs/[1-9][0-9]*/artifacts$" - }, - "relativePath": { - "type": "string", - "maxLength": 512, - "pattern": "^[A-Za-z0-9][A-Za-z0-9._-]{0,127}(?:/[A-Za-z0-9][A-Za-z0-9._-]{0,127}){0,7}$" - }, - "benchmark": { "enum": ["evmbench", "ultrafuzz-bench"] }, - "lane": { "enum": ["smoke", "full"] }, - "status": { "enum": ["succeeded", "genuine-task-failures", "failed"] }, - "targetIds": { - "type": "array", - "minItems": 1, - "maxItems": 2048, - "uniqueItems": true, - "items": { "$ref": "#/$defs/safeId" } - }, - "pair": { - "type": "object", - "additionalProperties": false, - "required": [ - "pair", - "provider", - "model_slug", - "bundle_path", - "unpack_path", - "eval_run_id", - "benchmark", - "lane", - "status", - "target_ids", - "executed_case_count", - "graded_case_count", - "publication_url" - ], - "properties": { - "pair": { "$ref": "#/$defs/safeLowerId" }, - "provider": { "enum": ["openai", "anthropic", "kimi", "deepseek", "openrouter"] }, - "model_slug": { "$ref": "#/$defs/safeLowerId" }, - "bundle_path": { "$ref": "#/$defs/relativePath" }, - "unpack_path": { "$ref": "#/$defs/safeLowerId" }, - "eval_run_id": { "$ref": "#/$defs/safeId" }, - "benchmark": { "$ref": "#/$defs/benchmark" }, - "lane": { "$ref": "#/$defs/lane" }, - "status": { "$ref": "#/$defs/status" }, - "target_ids": { "$ref": "#/$defs/targetIds" }, - "executed_case_count": { "$ref": "#/$defs/positiveInteger" }, - "graded_case_count": { "$ref": "#/$defs/positiveInteger" }, - "publication_url": { "$ref": "#/$defs/actionsArtifacts" } - } - } - } -} diff --git a/packages/evals/schema/eval-history-publication-generation.schema.json b/packages/evals/schema/eval-history-publication-generation.schema.json deleted file mode 100644 index 18570e5c9..000000000 --- a/packages/evals/schema/eval-history-publication-generation.schema.json +++ /dev/null @@ -1,91 +0,0 @@ -{ - "$schema": "https://json-schema.org/draft/2020-12/schema", - "$id": "urn:ultrafuzz:schema:evals:history-publication-generation:1", - "title": "Ultrafuzz eval-history publication generation v1", - "type": "object", - "additionalProperties": false, - "required": ["schema_version", "candidate_commit", "candidate_repository_url", "source_artifact", "runs"], - "properties": { - "schema_version": { "const": "ultrafuzz.eval-history-publication-generation.v1" }, - "candidate_commit": { "$ref": "#/$defs/fullCommit" }, - "candidate_repository_url": { "$ref": "#/$defs/githubRepository" }, - "source_artifact": { "$ref": "#/$defs/actionsRun" }, - "runs": { - "type": "array", - "minItems": 1, - "maxItems": 64, - "items": { "$ref": "#/$defs/run" } - } - }, - "$defs": { - "safeId": { - "type": "string", - "pattern": "^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$" - }, - "fullCommit": { - "type": "string", - "pattern": "^[0-9a-f]{40}$" - }, - "positiveInteger": { - "type": "integer", - "minimum": 1, - "maximum": 9007199254740991 - }, - "githubRepository": { - "type": "string", - "maxLength": 2048, - "pattern": "^https://github\\.com/[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+$" - }, - "actionsRun": { - "type": "string", - "maxLength": 2048, - "pattern": "^https://github\\.com/[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+/actions/runs/[1-9][0-9]*$" - }, - "actionsArtifacts": { - "type": "string", - "maxLength": 2048, - "pattern": "^https://github\\.com/[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+/actions/runs/[1-9][0-9]*/artifacts$" - }, - "relativePath": { - "type": "string", - "maxLength": 512, - "pattern": "^[A-Za-z0-9][A-Za-z0-9._-]{0,127}(?:/[A-Za-z0-9][A-Za-z0-9._-]{0,127}){0,7}$" - }, - "benchmark": { "enum": ["evmbench", "ultrafuzz-bench"] }, - "lane": { "enum": ["smoke", "full"] }, - "status": { "enum": ["succeeded", "genuine-task-failures", "failed"] }, - "targetIds": { - "type": "array", - "minItems": 1, - "maxItems": 2048, - "uniqueItems": true, - "items": { "$ref": "#/$defs/safeId" } - }, - "run": { - "type": "object", - "additionalProperties": false, - "required": [ - "eval_run_id", - "benchmark", - "lane", - "status", - "input_path", - "target_ids", - "executed_case_count", - "graded_case_count", - "publication_url" - ], - "properties": { - "eval_run_id": { "$ref": "#/$defs/safeId" }, - "benchmark": { "$ref": "#/$defs/benchmark" }, - "lane": { "$ref": "#/$defs/lane" }, - "status": { "$ref": "#/$defs/status" }, - "input_path": { "$ref": "#/$defs/relativePath" }, - "target_ids": { "$ref": "#/$defs/targetIds" }, - "executed_case_count": { "$ref": "#/$defs/positiveInteger" }, - "graded_case_count": { "$ref": "#/$defs/positiveInteger" }, - "publication_url": { "$ref": "#/$defs/actionsArtifacts" } - } - } - } -} diff --git a/packages/evals/schema/eval-publication-state.schema.json b/packages/evals/schema/eval-publication-state.schema.json deleted file mode 100644 index c90cab1bc..000000000 --- a/packages/evals/schema/eval-publication-state.schema.json +++ /dev/null @@ -1,5 +0,0 @@ -{ - "$schema": "https://json-schema.org/draft/2020-12/schema", - "$id": "urn:ultrafuzz:schema:evals:publication-state:1", - "$ref": "urn:ultrafuzz:schema:evals:common:3#/$defs/evalPublicationState" -} diff --git a/packages/evals/schema/telemetry-cursor.schema.json b/packages/evals/schema/telemetry-cursor.schema.json deleted file mode 100644 index 9554b20b4..000000000 --- a/packages/evals/schema/telemetry-cursor.schema.json +++ /dev/null @@ -1,5 +0,0 @@ -{ - "$schema": "https://json-schema.org/draft/2020-12/schema", - "$id": "urn:ultrafuzz:schema:evals:telemetry-cursor:1", - "$ref": "urn:ultrafuzz:schema:evals:common:3#/$defs/telemetryCursor" -} diff --git a/packages/evals/src/eval-durable.ts b/packages/evals/src/eval-durable.ts index 685107102..d78f66569 100644 --- a/packages/evals/src/eval-durable.ts +++ b/packages/evals/src/eval-durable.ts @@ -10,30 +10,25 @@ import { writeJsonDurable } from "@ultrafuzz/artifacts"; -import type { TelemetryCursorState } from "./node-telemetry.js"; import { EVAL_FINDING_SCORE_SCHEMA_ID, EVAL_MATRIX_SCHEMA_ID, - EVAL_PUBLICATION_STATE_SCHEMA_ID, EVAL_REVIEW_QUEUE_ITEM_SCHEMA_ID, EVAL_RUN_MANIFEST_SCHEMA_ID, EVAL_RUN_RECORD_SCHEMA_ID, EVAL_RUN_SUMMARY_SCHEMA_ID, EVAL_SCORE_SUMMARY_SCHEMA_ID, - EVAL_TELEMETRY_CURSOR_SCHEMA_ID, validateEvalJsonSchema } from "./eval-schema-registry.js"; import { assertEvalSemanticGateRegistry, executeEvalSchemaSemanticGates } from "./eval-semantic-gates.js"; import { EVAL_FINDING_SCORE_SCHEMA_VERSION, - EVAL_PUBLICATION_STATE_SCHEMA_VERSION, EVAL_REVIEW_QUEUE_ITEM_SCHEMA_VERSION, EVAL_RUN_SCHEMA_VERSION, EVAL_RUN_SUMMARY_SCHEMA_VERSION, EVAL_SCORE_SUMMARY_SCHEMA_VERSION, type EvalFindingScore, type EvalMatrixRow, - type EvalPublicationState, type EvalRunManifest, type EvalRunRecord, type EvalRunSummary, @@ -348,32 +343,6 @@ export function writeEvalScoreSummary(filePath: string, value: EvalScoreSummary) writeJsonDurable(filePath, parseEvalScoreSummary(value, filePath)); } -export function parseEvalPublicationState(value: unknown, source = "publication-state.json"): EvalPublicationState { - assertVersion(value, "schema_version", EVAL_PUBLICATION_STATE_SCHEMA_VERSION, source); - return validate(EVAL_PUBLICATION_STATE_SCHEMA_ID, value, source); -} - -export function readEvalPublicationState(filePath: string): EvalPublicationState { - return parseEvalPublicationState(readStrictJsonDocument(filePath), filePath); -} - -export function writeEvalPublicationState(filePath: string, value: EvalPublicationState): void { - writeJsonDurable(filePath, parseEvalPublicationState(value, filePath)); -} - -export function parseTelemetryCursor(value: unknown, source = "telemetry cursor"): TelemetryCursorState { - assertVersion(value, "schemaVersion", "ultrafuzz.eval.telemetry-cursor.v1", source); - return validate(EVAL_TELEMETRY_CURSOR_SCHEMA_ID, value, source); -} - -export function readTelemetryCursor(filePath: string): TelemetryCursorState { - return parseTelemetryCursor(readStrictJsonDocument(filePath), filePath); -} - -export function writeTelemetryCursor(filePath: string, value: TelemetryCursorState): void { - writeJsonDurable(filePath, parseTelemetryCursor(value, filePath)); -} - export function readStrictJsonDocument(filePath: string): unknown { let bytes: Buffer; try { diff --git a/packages/evals/src/eval-schema-registry.ts b/packages/evals/src/eval-schema-registry.ts index e0a0e090c..e481317bb 100644 --- a/packages/evals/src/eval-schema-registry.ts +++ b/packages/evals/src/eval-schema-registry.ts @@ -23,10 +23,6 @@ export const EVAL_SUITE_SCHEMA_ID = "urn:ultrafuzz:schema:evals:suite:2" as cons export const EVAL_ADJUDICATION_HANDOFF_SCHEMA_ID = "urn:ultrafuzz:schema:evals:adjudication-handoff:1" as const; export const EVAL_FINDING_MANIFEST_SCHEMA_ID = "urn:ultrafuzz:schema:evals:finding-manifest:1" as const; export const EVAL_GROUND_TRUTH_SCHEMA_ID = "urn:ultrafuzz:schema:evals:ground-truth:1" as const; -export const EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_ID = - "urn:ultrafuzz:schema:evals:history-automatic-publication-plan:1" as const; -export const EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_ID = - "urn:ultrafuzz:schema:evals:history-publication-generation:1" as const; export const EVAL_HISTORY_SCHEMA_ID = "urn:ultrafuzz:schema:evals:history:2" as const; export const EVAL_INSTANCE_CLUSTERS_SCHEMA_ID = "urn:ultrafuzz:schema:evals:instance-clusters:1" as const; export const EVAL_GROUND_TRUTH_CREDITS_SCHEMA_ID = "urn:ultrafuzz:schema:evals:ground-truth-credits:1" as const; @@ -49,9 +45,7 @@ export const EVAL_SCORE_SUMMARY_SCHEMA_ID = "urn:ultrafuzz:schema:evals:score-su export const EVAL_RECOVERY_EQUIVALENCE_SCHEMA_ID = "urn:ultrafuzz:schema:evals:recovery-equivalence:1" as const; export const EVAL_STATUS_SCHEMA_ID = "urn:ultrafuzz:schema:evals:status:1" as const; export const EVAL_REVIEW_QUEUE_ITEM_SCHEMA_ID = "urn:ultrafuzz:schema:evals:review-queue-item:2" as const; -export const EVAL_PUBLICATION_STATE_SCHEMA_ID = "urn:ultrafuzz:schema:evals:publication-state:1" as const; export const EVAL_PUBLIC_DIAGNOSTICS_SCHEMA_ID = "urn:ultrafuzz:schema:evals:public-eval-diagnostics:2" as const; -export const EVAL_TELEMETRY_CURSOR_SCHEMA_ID = "urn:ultrafuzz:schema:evals:telemetry-cursor:1" as const; const MAX_EVAL_SCHEMA_BYTES = 2 * 1024 * 1024; @@ -125,16 +119,6 @@ export const EVAL_SCHEMA_METADATA: Readonly> zodParser: "evalHistoryZodSchema", semanticGates: ["eval-history-integrity"] }, - "eval-history-automatic-publication-plan.schema.json": { - role: "runtime-state", - typescriptExport: "evalHistoryAutomaticPublicationPlanJsonSchema", - semanticGates: ["eval-history-automatic-publication-plan-integrity"] - }, - "eval-history-publication-generation.schema.json": { - role: "runtime-state", - typescriptExport: "evalHistoryPublicationGenerationJsonSchema", - semanticGates: ["eval-history-publication-generation-integrity"] - }, "eval-matrix.schema.json": { role: "runtime-state", typescriptExport: "evalMatrixJsonSchema", @@ -145,11 +129,6 @@ export const EVAL_SCHEMA_METADATA: Readonly> typescriptExport: "evalLlmJudgeResultJsonSchema", semanticGates: [] }, - "eval-publication-state.schema.json": { - role: "runtime-state", - typescriptExport: "evalPublicationStateJsonSchema", - semanticGates: [] - }, "eval-recovery-equivalence.schema.json": { role: "runtime-state", typescriptExport: "evalRecoveryEquivalenceJsonSchema", @@ -226,11 +205,6 @@ export const EVAL_SCHEMA_METADATA: Readonly> role: "runtime-state", typescriptExport: "evalInstanceClustersJsonSchema", semanticGates: ["eval-instance-clusters-identity-joins"] - }, - "telemetry-cursor.schema.json": { - role: "runtime-state", - typescriptExport: "evalTelemetryCursorJsonSchema", - semanticGates: [] } }); @@ -266,12 +240,6 @@ export const evalBenchmarkProvenanceJsonSchema = loadSchemaDocument("benchmark-p export const evalBenchmarkSourceManifestJsonSchema = loadSchemaDocument("benchmark-source-manifest.schema.json"); export const evalFindingScoreJsonSchema = loadSchemaDocument("eval-finding-score.schema.json"); export const evalGroundTruthJsonSchema = loadSchemaDocument("eval-ground-truth.schema.json"); -export const evalHistoryAutomaticPublicationPlanJsonSchema = loadSchemaDocument( - "eval-history-automatic-publication-plan.schema.json" -); -export const evalHistoryPublicationGenerationJsonSchema = loadSchemaDocument( - "eval-history-publication-generation.schema.json" -); export const evalHistoryJsonSchema = loadSchemaDocument("eval-history.schema.json"); export const evalFindingManifestJsonSchema = loadSchemaDocument("finding-manifest.schema.json"); export const evalGroundTruthCreditsJsonSchema = loadSchemaDocument("ground-truth-credits.schema.json"); @@ -279,7 +247,6 @@ export const evalInstanceClustersJsonSchema = loadSchemaDocument("instance-clust export const evalLlmJudgeResultJsonSchema = loadSchemaDocument("eval-llm-judge-result.schema.json"); export const evalMatrixJsonSchema = loadSchemaDocument("eval-matrix.schema.json"); export const evalPublicDiagnosticsJsonSchema = loadSchemaDocument("eval-public-diagnostics.schema.json"); -export const evalPublicationStateJsonSchema = loadSchemaDocument("eval-publication-state.schema.json"); export const evalRecoveryEquivalenceJsonSchema = loadSchemaDocument("eval-recovery-equivalence.schema.json"); export const evalReviewQueueItemJsonSchema = loadSchemaDocument("eval-review-queue-item.schema.json"); export const evalRunManifestJsonSchema = loadSchemaDocument("eval-run-manifest.schema.json"); @@ -289,7 +256,6 @@ export const evalScoreSummaryJsonSchema = loadSchemaDocument("eval-score-summary export const evalStatusJsonSchema = loadSchemaDocument("eval-status.schema.json"); export const evalSuiteJsonSchema = loadSchemaDocument("eval-suite.schema.json"); export const evalEvmbenchCohortJsonSchema = loadSchemaDocument("evmbench-cohort.schema.json"); -export const evalTelemetryCursorJsonSchema = loadSchemaDocument("telemetry-cursor.schema.json"); export const EVAL_SCHEMA_EXPORTS = Object.freeze({ evalAdjudicationHandoffJsonSchema, @@ -301,8 +267,6 @@ export const EVAL_SCHEMA_EXPORTS = Object.freeze({ evalCommonJsonSchema, evalFindingScoreJsonSchema, evalGroundTruthJsonSchema, - evalHistoryAutomaticPublicationPlanJsonSchema, - evalHistoryPublicationGenerationJsonSchema, evalHistoryJsonSchema, evalFindingManifestJsonSchema, evalGroundTruthCreditsJsonSchema, @@ -310,7 +274,6 @@ export const EVAL_SCHEMA_EXPORTS = Object.freeze({ evalLlmJudgeResultJsonSchema, evalMatrixJsonSchema, evalPublicDiagnosticsJsonSchema, - evalPublicationStateJsonSchema, evalRecoveryEquivalenceJsonSchema, evalReviewQueueItemJsonSchema, evalRunManifestJsonSchema, @@ -319,8 +282,7 @@ export const EVAL_SCHEMA_EXPORTS = Object.freeze({ evalScoreSummaryJsonSchema, evalStatusJsonSchema, evalSuiteJsonSchema, - evalEvmbenchCohortJsonSchema, - evalTelemetryCursorJsonSchema + evalEvmbenchCohortJsonSchema }); const schemaExportsByFilename: Readonly>>> = Object.freeze({ @@ -333,12 +295,9 @@ const schemaExportsByFilename: Readonly> = Object.freeze({ "eval-ground-truth-integrity": groundTruthIntegrity, "eval-ground-truth-credits-identity-joins": groundTruthCreditsIdentityJoins, "eval-history-integrity": historyIntegrity, - "eval-history-automatic-publication-plan-integrity": historyAutomaticPublicationPlanIntegrity, - "eval-history-publication-generation-integrity": historyPublicationGenerationIntegrity, "eval-instance-clusters-identity-joins": instanceClustersIdentityJoins, "eval-matrix-identity-joins": matrixIdentityJoins, "eval-public-diagnostics-consistency": publicDiagnosticsConsistency, @@ -119,8 +109,6 @@ const gatesBySchema: Readonly> = Object.freeze [EVAL_GROUND_TRUTH_SCHEMA_ID]: ["eval-ground-truth-integrity"], [EVAL_GROUND_TRUTH_CREDITS_SCHEMA_ID]: ["eval-ground-truth-credits-identity-joins"], [EVAL_HISTORY_SCHEMA_ID]: ["eval-history-integrity"], - [EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_ID]: ["eval-history-automatic-publication-plan-integrity"], - [EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_ID]: ["eval-history-publication-generation-integrity"], [EVAL_INSTANCE_CLUSTERS_SCHEMA_ID]: ["eval-instance-clusters-identity-joins"], [EVAL_MATRIX_SCHEMA_ID]: ["eval-matrix-identity-joins"], [EVAL_PUBLIC_DIAGNOSTICS_SCHEMA_ID]: ["eval-public-diagnostics-consistency"], @@ -379,18 +367,6 @@ function historyIntegrity(value: unknown): EvalSemanticGateIssue[] { ); } -function historyAutomaticPublicationPlanIntegrity(value: unknown): EvalSemanticGateIssue[] { - return evalHistoryAutomaticPublicationPlanSemanticIssues(value as EvalHistoryAutomaticPublicationPlan).map((entry) => - issue("eval-history-automatic-publication-plan-integrity", entry.path, entry.message) - ); -} - -function historyPublicationGenerationIntegrity(value: unknown): EvalSemanticGateIssue[] { - return evalHistoryPublicationGenerationSemanticIssues(value as EvalHistoryPublicationGeneration).map((entry) => - issue("eval-history-publication-generation-integrity", entry.path, entry.message) - ); -} - function findingManifestIdentityJoins(value: unknown): EvalSemanticGateIssue[] { const manifest = value as BenchmarkFindingManifest; const gate = "eval-finding-manifest-identity-joins"; diff --git a/packages/evals/src/expansion.ts b/packages/evals/src/expansion.ts index d4e33f89f..82d91051a 100644 --- a/packages/evals/src/expansion.ts +++ b/packages/evals/src/expansion.ts @@ -9,7 +9,6 @@ import { readRunMetadataDocument, type GoalPlan, type NodeState, - type NodeStatus, type RunState, type UsageLedgerEntry } from "@ultrafuzz/artifacts"; @@ -35,25 +34,15 @@ import type { */ export const MAX_EVAL_EXPANSION_NODE_IDS = 256; -/** - * Key a run is expected to record on a generated node's provenance to name the - * node that generated it. #183 asks for "source node IDs" per run record; this - * is the name that request is read under. - */ -export const EVAL_EXPANSION_SOURCE_NODE_KEY = "source_node_id"; - /** * Node-level view of one row, derived only from the run's own durable evidence. * * This exists because `EvalRunRecord` carried nothing below the run: it had a * single `workflow` lifecycle and no node counts, no identifiers and no * concurrency, so a fan-out was indistinguishable from one opaque agent node. - * The alternative channel does not work either -- `node_telemetry` reaches only - * `this.input.reporters`, while the public worker uses local reporting and has - * no external reporters. * * Everything here comes from `state.json` and `graph.json`, which every run - * writes, so it needs no reporter, no provider and no network. + * writes, so it needs no provider and no network. */ export function evalRunExpansion(input: { runRoot: string; state: RunState }): EvalRunExpansion { const nodes = Object.values(input.state.nodes); @@ -511,7 +500,3 @@ function completeness(reason: EvalExpansionReason | undefined): EvalExpansionCom function compareIds(left: string, right: string): number { return left < right ? -1 : left > right ? 1 : 0; } - -export function isEvalNodeStatus(value: string): value is NodeStatus { - return (NODE_STATE_STATUSES as readonly string[]).includes(value); -} diff --git a/packages/evals/src/history-publication.ts b/packages/evals/src/history-publication.ts deleted file mode 100644 index b29ed4afb..000000000 --- a/packages/evals/src/history-publication.ts +++ /dev/null @@ -1,360 +0,0 @@ -import { isDeepStrictEqual } from "node:util"; - -import { parseStrictJsonBytes, readRegularFileSnapshot } from "@ultrafuzz/artifacts"; - -import { - EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_ID, - EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_ID, - validateEvalJsonSchema -} from "./eval-schema-registry.js"; -import { EvalError } from "./utils.js"; - -export const EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_VERSION = - "ultrafuzz.eval-history-automatic-publication-plan.v1" as const; -export const EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_VERSION = - "ultrafuzz.eval-history-publication-generation.v1" as const; - -const MAX_HISTORY_PUBLICATION_DOCUMENT_BYTES = 1024 * 1024; -export const EVAL_HISTORY_PUBLICATION_HANDOFF_GATE = "eval-history-publication-plan-generation-join" as const; - -export type EvalHistoryPublicationBenchmark = "evmbench" | "ultrafuzz-bench"; -export type EvalHistoryPublicationLane = "smoke" | "full"; -export type EvalHistoryPublicationStatus = "succeeded" | "genuine-task-failures" | "failed"; -export type EvalHistoryPublicationProvider = "openai" | "anthropic" | "kimi" | "deepseek" | "openrouter"; - -export interface EvalHistoryAutomaticPublicationPair { - pair: string; - provider: EvalHistoryPublicationProvider; - model_slug: string; - bundle_path: string; - unpack_path: string; - eval_run_id: string; - benchmark: EvalHistoryPublicationBenchmark; - lane: EvalHistoryPublicationLane; - status: EvalHistoryPublicationStatus; - target_ids: string[]; - executed_case_count: number; - graded_case_count: number; - publication_url: string; -} - -export interface EvalHistoryAutomaticPublicationPlan { - schema_version: typeof EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_VERSION; - candidate_commit: string; - candidate_repository_url: string; - source_artifact: string; - producer_run_id: string; - producer_run_attempt: string; - mode: EvalHistoryPublicationLane; - benchmark: EvalHistoryPublicationBenchmark; - pairs: EvalHistoryAutomaticPublicationPair[]; -} - -export interface EvalHistoryPublicationRun { - eval_run_id: string; - benchmark: EvalHistoryPublicationBenchmark; - lane: EvalHistoryPublicationLane; - status: EvalHistoryPublicationStatus; - input_path: string; - target_ids: string[]; - executed_case_count: number; - graded_case_count: number; - publication_url: string; -} - -export interface EvalHistoryPublicationGeneration { - schema_version: typeof EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_VERSION; - candidate_commit: string; - candidate_repository_url: string; - source_artifact: string; - runs: EvalHistoryPublicationRun[]; -} - -export interface EvalHistoryPublicationSemanticIssue { - path: string; - message: string; -} - -export interface EvalHistoryPublicationHandoffIssue extends EvalHistoryPublicationSemanticIssue { - gate: typeof EVAL_HISTORY_PUBLICATION_HANDOFF_GATE; -} - -export function parseEvalHistoryAutomaticPublicationPlan( - value: unknown, - source = "automatic eval-history publication plan" -): EvalHistoryAutomaticPublicationPlan { - const document = validateShape( - EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_ID, - value, - source, - "EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_INVALID" - ); - const issues = evalHistoryAutomaticPublicationPlanSemanticIssues(document); - if (issues.length > 0) { - throw new EvalError("EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_INVALID", `${source} failed semantic validation`, { - issues - }); - } - return document; -} - -export function readEvalHistoryAutomaticPublicationPlan(filePath: string): EvalHistoryAutomaticPublicationPlan { - return parseEvalHistoryAutomaticPublicationPlan( - readPublicationDocument(filePath, "automatic eval-history publication plan"), - filePath - ); -} - -export function parseEvalHistoryPublicationGeneration( - value: unknown, - source = "eval-history publication generation" -): EvalHistoryPublicationGeneration { - const document = validateShape( - EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_ID, - value, - source, - "EVAL_HISTORY_PUBLICATION_GENERATION_INVALID" - ); - const issues = evalHistoryPublicationGenerationSemanticIssues(document); - if (issues.length > 0) { - throw new EvalError("EVAL_HISTORY_PUBLICATION_GENERATION_INVALID", `${source} failed semantic validation`, { - issues - }); - } - return document; -} - -export function readEvalHistoryPublicationGeneration(filePath: string): EvalHistoryPublicationGeneration { - return parseEvalHistoryPublicationGeneration( - readPublicationDocument(filePath, "eval-history publication generation"), - filePath - ); -} - -export function evalHistoryAutomaticPublicationPlanSemanticIssues( - plan: EvalHistoryAutomaticPublicationPlan -): EvalHistoryPublicationSemanticIssue[] { - const issues: EvalHistoryPublicationSemanticIssue[] = []; - const expectedBenchmark = plan.mode === "smoke" ? "ultrafuzz-bench" : "evmbench"; - if (plan.benchmark !== expectedBenchmark) { - issues.push({ path: "$.benchmark", message: `${plan.mode} publication must use ${expectedBenchmark}` }); - } - const expectedSourceArtifact = `${plan.candidate_repository_url}/actions/runs/${plan.producer_run_id}`; - if (plan.source_artifact !== expectedSourceArtifact) { - issues.push({ - path: "$.source_artifact", - message: "must identify producer_run_id in candidate_repository_url" - }); - } - const expectedProviders: readonly EvalHistoryPublicationProvider[] = - plan.mode === "full" ? ["openai", "anthropic", "kimi", "deepseek"] : [plan.pairs[0]?.provider ?? "openai"]; - if ( - plan.pairs.length !== expectedProviders.length || - plan.pairs.some((pair, index) => pair.provider !== expectedProviders[index]) - ) { - issues.push({ - path: "$.pairs", - message: - plan.mode === "full" - ? "full publication must contain the ordered openai, anthropic, kimi, and deepseek pairs" - : "smoke publication must contain exactly one known provider pair" - }); - } - const identities = new Set(); - const modelSlugs = new Set(); - const bundlePaths = new Set(); - const unpackPaths = new Set(); - const evalRunIds = new Set(); - const expectedPublicationUrl = `${plan.source_artifact}/artifacts`; - const expectedTargetIds = JSON.stringify(plan.pairs[0]?.target_ids ?? []); - plan.pairs.forEach((pair, index) => { - uniqueIssue(issues, identities, pair.pair, `$.pairs[${index}].pair`, "pair ID"); - uniqueIssue(issues, modelSlugs, pair.model_slug, `$.pairs[${index}].model_slug`, "model slug"); - uniqueIssue(issues, bundlePaths, pair.bundle_path, `$.pairs[${index}].bundle_path`, "bundle path"); - uniqueIssue(issues, unpackPaths, pair.unpack_path, `$.pairs[${index}].unpack_path`, "unpack path"); - uniqueIssue(issues, evalRunIds, pair.eval_run_id, `$.pairs[${index}].eval_run_id`, "eval run ID"); - if (pair.benchmark !== plan.benchmark || pair.lane !== plan.mode) { - issues.push({ - path: `$.pairs[${index}]`, - message: "pair benchmark and lane must equal the plan benchmark and mode" - }); - } - if (pair.pair !== `${pair.benchmark}-${pair.model_slug}`) { - issues.push({ path: `$.pairs[${index}].pair`, message: "must equal benchmark plus model_slug" }); - } - if (pair.unpack_path !== pair.pair) { - issues.push({ path: `$.pairs[${index}].unpack_path`, message: "must equal pair" }); - } - if (pair.bundle_path !== `${pair.pair}/${pair.model_slug}/public-results.json`) { - issues.push({ - path: `$.pairs[${index}].bundle_path`, - message: "must be the canonical public-results path for pair and model_slug" - }); - } - if (pair.publication_url !== expectedPublicationUrl) { - issues.push({ - path: `$.pairs[${index}].publication_url`, - message: "must identify the producer run artifacts" - }); - } - if (JSON.stringify(pair.target_ids) !== expectedTargetIds) { - issues.push({ path: `$.pairs[${index}].target_ids`, message: "all pairs must name the same ordered targets" }); - } - caseCountIssues(issues, pair, `$.pairs[${index}]`); - }); - return issues; -} - -export function evalHistoryPublicationGenerationSemanticIssues( - generation: EvalHistoryPublicationGeneration -): EvalHistoryPublicationSemanticIssue[] { - const issues: EvalHistoryPublicationSemanticIssue[] = []; - const expectedSourcePrefix = `${generation.candidate_repository_url}/actions/runs/`; - if (!generation.source_artifact.startsWith(expectedSourcePrefix)) { - issues.push({ - path: "$.source_artifact", - message: "must identify a run in candidate_repository_url" - }); - } - const expectedPublicationUrl = `${generation.source_artifact}/artifacts`; - const evalRunIds = new Set(); - const inputPaths = new Set(); - const first = generation.runs[0]; - const expectedTargetIds = JSON.stringify(first?.target_ids ?? []); - generation.runs.forEach((run, index) => { - uniqueIssue(issues, evalRunIds, run.eval_run_id, `$.runs[${index}].eval_run_id`, "eval run ID"); - uniqueIssue(issues, inputPaths, run.input_path, `$.runs[${index}].input_path`, "input path"); - const expectedBenchmark = run.lane === "smoke" ? "ultrafuzz-bench" : "evmbench"; - if (run.benchmark !== expectedBenchmark) { - issues.push({ - path: `$.runs[${index}].benchmark`, - message: `${run.lane} publication must use ${expectedBenchmark}` - }); - } - if (first !== undefined && (run.benchmark !== first.benchmark || run.lane !== first.lane)) { - issues.push({ path: `$.runs[${index}]`, message: "all runs must share one benchmark and lane" }); - } - if (JSON.stringify(run.target_ids) !== expectedTargetIds) { - issues.push({ path: `$.runs[${index}].target_ids`, message: "all runs must name the same ordered targets" }); - } - if (run.publication_url !== expectedPublicationUrl) { - issues.push({ - path: `$.runs[${index}].publication_url`, - message: "must identify the source_artifact run artifacts" - }); - } - caseCountIssues(issues, run, `$.runs[${index}]`); - }); - return issues; -} - -export function evalHistoryPublicationHandoffIssues( - plan: EvalHistoryAutomaticPublicationPlan, - generation: EvalHistoryPublicationGeneration -): EvalHistoryPublicationHandoffIssue[] { - const issues: EvalHistoryPublicationHandoffIssue[] = []; - for (const field of ["candidate_commit", "candidate_repository_url", "source_artifact"] as const) { - if (plan[field] !== generation[field]) { - issues.push({ - gate: EVAL_HISTORY_PUBLICATION_HANDOFF_GATE, - path: `$.generation.${field}`, - message: `must equal plan.${field}` - }); - } - } - if (generation.runs.length !== plan.pairs.length) { - issues.push({ - gate: EVAL_HISTORY_PUBLICATION_HANDOFF_GATE, - path: "$.generation.runs", - message: "must contain exactly one run for each plan pair" - }); - } - plan.pairs.forEach((pair, index) => { - const expected: EvalHistoryPublicationRun = { - eval_run_id: pair.eval_run_id, - benchmark: pair.benchmark, - lane: pair.lane, - status: pair.status, - input_path: `${pair.unpack_path}/eval`, - target_ids: pair.target_ids, - executed_case_count: pair.executed_case_count, - graded_case_count: pair.graded_case_count, - publication_url: pair.publication_url - }; - if (!isDeepStrictEqual(generation.runs[index], expected)) { - issues.push({ - gate: EVAL_HISTORY_PUBLICATION_HANDOFF_GATE, - path: `$.generation.runs[${index}]`, - message: `must equal the canonical projection of plan.pairs[${index}]` - }); - } - }); - return issues; -} - -export function assertEvalHistoryPublicationHandoff( - plan: EvalHistoryAutomaticPublicationPlan, - generation: EvalHistoryPublicationGeneration -): void { - const issues = evalHistoryPublicationHandoffIssues(plan, generation); - if (issues.length > 0) { - throw new EvalError("EVAL_HISTORY_PUBLICATION_HANDOFF_INVALID", "publication plan and generation do not join", { - gate: EVAL_HISTORY_PUBLICATION_HANDOFF_GATE, - issues - }); - } -} - -function validateShape(schemaId: string, value: unknown, source: string, code: string): DocumentType { - const validation = validateEvalJsonSchema(schemaId, value); - if (!validation.ok) { - throw new EvalError(code, `${source} failed schema validation`, { - schema_id: schemaId, - issues: validation.issues, - truncated: validation.truncated - }); - } - return value as DocumentType; -} - -function readPublicationDocument(filePath: string, label: string): unknown { - try { - return parseStrictJsonBytes(readRegularFileSnapshot(filePath, MAX_HISTORY_PUBLICATION_DOCUMENT_BYTES), { - maxBytes: MAX_HISTORY_PUBLICATION_DOCUMENT_BYTES, - maxDepth: 32, - maxItems: 50_000, - maxProperties: 50_000 - }); - } catch (error) { - const reason = error instanceof Error ? error.message : String(error); - throw new EvalError( - "EVAL_HISTORY_PUBLICATION_DOCUMENT_INVALID", - `${label} is unreadable or invalid: ${filePath}: ${reason}`, - { path: filePath, reason } - ); - } -} - -function uniqueIssue( - issues: EvalHistoryPublicationSemanticIssue[], - seen: Set, - value: string, - path: string, - label: string -): void { - if (seen.has(value)) issues.push({ path, message: `${label} must be unique` }); - seen.add(value); -} - -function caseCountIssues( - issues: EvalHistoryPublicationSemanticIssue[], - value: { target_ids: string[]; executed_case_count: number; graded_case_count: number }, - path: string -): void { - if (value.graded_case_count > value.executed_case_count) { - issues.push({ path: `${path}.graded_case_count`, message: "must not exceed executed_case_count" }); - } - if (value.target_ids.length > value.executed_case_count || value.target_ids.length > value.graded_case_count) { - issues.push({ path, message: "case counts must cover every target" }); - } -} diff --git a/packages/evals/src/history.ts b/packages/evals/src/history.ts index 0c596dbf9..86334b233 100644 --- a/packages/evals/src/history.ts +++ b/packages/evals/src/history.ts @@ -47,10 +47,6 @@ export const EVAL_HISTORY_OBSERVATION_SCHEMA_VERSION = "ultrafuzz.eval.history.o const EVAL_HISTORY_PUBLIC_BUNDLE_FILE = "public-results.json"; const EVAL_HISTORY_PUBLIC_REPORT_FILES = ["report.md", "report.json"] as const; const MAX_EVAL_HISTORY_BYTES = 64 * 1024 * 1024; -const MILLISECONDS_PER_DAY = 86_400_000; -// The rendered age must carry more precision than the rendered maximum, so an age -// barely over the budget cannot print as the budget itself. -const EVAL_HISTORY_AGE_FRACTION_DIGITS = 4; export type EvalHistoryBenchmark = "evmbench" | "ultrafuzz-bench"; export type EvalHistoryLane = BenchmarkLaneName; @@ -261,20 +257,6 @@ export const evalHistoryZodSchema = z.strictObject({ observations: z.array(observationSchema).max(100_000) }); -export const EVAL_HISTORY_CHARTS = [ - { file: "precision.svg", metric: "precision", title: "Precision", ratio: true }, - { file: "recall.svg", metric: "recall", title: "Recall", ratio: true }, - { file: "f1.svg", metric: "f1", title: "F1", ratio: true }, - { - file: "cumulative-unique-true-positives.svg", - metric: "cumulative_unique_true_positives", - title: "Cumulative unique true positives", - ratio: false - }, - { file: "wall-clock-time.svg", metric: "wall_clock_seconds", title: "Wall-clock time (seconds)", ratio: false }, - { file: "cost.svg", metric: "cost_usd", title: "Cost (USD)", ratio: false } -] as const; - export const EVAL_HISTORY_OVERVIEW_FILES = ["latest-summary.svg", "quality.svg", "performance-cost.svg"] as const; // Luna's repository history has a non-overlapping cost regime beginning with @@ -285,8 +267,6 @@ export const EVAL_HISTORY_PERFORMANCE_COST_MODEL_CUTOFFS: Readonly newest.milliseconds) { - newest = { timestamp: observation.run_timestamp, milliseconds }; - } - } - if (newest === undefined) { - throw new EvalError( - "EVAL_HISTORY_EMPTY", - `eval history has no observation to age against the requested ${format(input.maxAgeDays)} day maximum`, - { max_age_days: input.maxAgeDays } - ); - } - // A future-dated observation would otherwise yield a negative age that passes every - // maximum, so one bad producer clock would silence this assertion permanently. - if (newest.milliseconds > now) { - const nowTimestamp = new Date(now).toISOString(); - throw new EvalError( - "EVAL_HISTORY_FUTURE_DATED", - `newest eval history observation ran at ${newest.timestamp}, which is in the future at ${nowTimestamp}, so its age cannot be measured against the requested ${format(input.maxAgeDays)} day maximum`, - { newest_run_timestamp: newest.timestamp, now: nowTimestamp, max_age_days: input.maxAgeDays } - ); - } - const ageDays = (now - newest.milliseconds) / MILLISECONDS_PER_DAY; - if (ageDays > input.maxAgeDays) { - throw new EvalError( - "EVAL_HISTORY_STALE", - `newest eval history observation ran at ${newest.timestamp}, ${format(ageDays, EVAL_HISTORY_AGE_FRACTION_DIGITS)} days ago, exceeding the requested ${format(input.maxAgeDays)} day maximum`, - { newest_run_timestamp: newest.timestamp, age_days: ageDays, max_age_days: input.maxAgeDays } - ); - } -} - export interface EvalHistoryGenerationInput { benchmark: EvalHistoryBenchmark; lane: EvalHistoryLane; @@ -2298,10 +2231,7 @@ export function renderEvalHistoryCharts(history: EvalHistory): Map [chart.file, renderChart(observations, chart.metric, chart.title, chart.ratio)] as const - ) + [EVAL_HISTORY_OVERVIEW_FILES[2], renderEvalPerformanceCostChart(aggregates)] ]); } @@ -2337,345 +2267,10 @@ export function formatEvalHistoryJson(history: EvalHistory): string { return `${JSON.stringify(history, null, 2).replace(SINGLE_ITEM_JSON_PRIMITIVE_ARRAY, "[$1]")}\n`; } -// Ordered series-identity fields. Benchmark lineage fields such as cohort and -// execution policy are intentionally excluded from line identity: those changes -// are rendered as vertical markers, while the metric lines keep tracking the -// same benchmark target/model over time. -const SERIES_CONTEXT_FIELDS = ["benchmark", "lane", "model", "reasoning"] as const; - -interface SeriesFields { - benchmark: string; - lane: string; - model: string; - reasoning: string; - cohort: string; - policy: string; - target: string; -} - -interface ChartPoint { - seriesKey: string; - fields: SeriesFields; - timestamp: string; - commit: string; - repositoryUrl: string; - value: number | null; - completeness: EvalHistoryCompleteness; - availableCount: number; - expectedCount: number; -} - -interface ChartColumn { - key: string; - timestamp: string; - commit: string; - repositoryUrl: string; -} - -interface LineageMarker { - columnKey: string; - label: string; - title: string; -} - -function chartPoints(observations: EvalHistoryObservation[], metric: ChartMetric): ChartPoint[] { - const groups = new Map(); - for (const observation of observations) { - const key = [ - observation.benchmark, - observation.lane, - observation.model, - observation.reasoning_effort, - observation.cohort_fingerprint, - observation.execution_policy_fingerprint, - observation.target, - observation.run_timestamp, - observation.candidate_commit - ].join("\u0000"); - groups.set(key, [...(groups.get(key) ?? []), observation]); - } - return [...groups.values()] - .map((group) => { - const first = group[0]!; - const fields: SeriesFields = { - benchmark: first.benchmark, - lane: first.lane, - model: first.model, - reasoning: first.reasoning_effort, - cohort: `cohort-${shortFingerprint(first.cohort_fingerprint)}`, - policy: `policy-${shortFingerprint(first.execution_policy_fingerprint)}`, - target: first.target - }; - const metricValue = aggregateChartMetric(group, metric); - return { - seriesKey: [...SERIES_CONTEXT_FIELDS.map((field) => fields[field]), fields.target].join(" "), - fields, - timestamp: first.run_timestamp, - commit: first.candidate_commit, - repositoryUrl: first.candidate_repository_url, - ...metricValue - }; - }) - .sort( - (left, right) => - compareText(left.seriesKey, right.seriesKey) || - compareText(left.timestamp, right.timestamp) || - compareText(left.commit, right.commit) - ); -} - -function chartColumnKey(point: Pick): string { - return [point.timestamp, point.commit].join("\u0000"); -} - -function chartColumns(points: ChartPoint[]): ChartColumn[] { - const byKey = new Map(); - for (const point of points) { - const key = chartColumnKey(point); - if (byKey.has(key)) continue; - byKey.set(key, { - key, - timestamp: point.timestamp, - commit: point.commit, - repositoryUrl: point.repositoryUrl - }); - } - return [...byKey.values()].sort( - (left, right) => compareText(left.timestamp, right.timestamp) || compareText(left.commit, right.commit) - ); -} - -function chartLineageMarkers(points: ChartPoint[]): LineageMarker[] { - const earliestByLineage = new Map(); - for (const point of [...points].sort( - (left, right) => compareText(left.timestamp, right.timestamp) || compareText(left.commit, right.commit) - )) { - const lineageKey = [point.fields.benchmark, point.fields.lane, point.fields.cohort, point.fields.policy].join( - "\u0000" - ); - if (!earliestByLineage.has(lineageKey)) earliestByLineage.set(lineageKey, point); - } - return [...earliestByLineage.values()].map((point) => ({ - columnKey: chartColumnKey(point), - label: point.fields.cohort, - title: `${point.fields.benchmark} ${point.fields.lane} ${point.fields.cohort} ${point.fields.policy}` - })); -} - function shortFingerprint(value: string): string { return value.replace(/^sha256:/u, "").slice(0, 8); } -function aggregateChartMetric( - observations: EvalHistoryObservation[], - metric: ChartMetric -): { - value: number | null; - completeness: EvalHistoryCompleteness; - availableCount: number; - expectedCount: number; -} { - if (metric === "cumulative_unique_true_positives") { - const perTarget = new Map(); - for (const observation of observations) { - perTarget.set( - observation.target, - Math.max(perTarget.get(observation.target) ?? 0, observation.cumulative_unique_true_positives) - ); - } - return { - value: [...perTarget.values()].reduce((sum, value) => sum + value, 0), - completeness: { status: "complete", reasons: [] }, - availableCount: observations.length, - expectedCount: observations.length - }; - } - if (metric === "wall_clock_seconds" || metric === "cost_usd") { - const completenessField = metric === "wall_clock_seconds" ? "wall_clock_completeness" : "cost_completeness"; - return aggregateCompletenessValues( - observations.map((observation) => ({ - value: observation[metric], - completeness: observation[completenessField] - })), - (values) => round(values.reduce((sum, value) => sum + value, 0)), - { preservePartialWithoutValue: true } - ); - } - return { - value: mean(observations.map((observation) => observation[metric])), - completeness: { status: "complete", reasons: [] }, - availableCount: observations.length, - expectedCount: observations.length - }; -} - -function renderChart( - observations: EvalHistoryObservation[], - metric: ChartMetric, - title: string, - ratioMetric: boolean -): string { - // Sized to be read at (or near) the README's full content width, one chart - // per row — a two-up layout would halve this and shrink the text again. - const width = 960; - const left = 70; - const right = 30; - const top = 104; - const plotWidth = width - left - right; - const plotHeight = 300; - const plotBottom = top + plotHeight; - const dateLabelBottom = plotBottom + 116; - const legendTop = dateLabelBottom + 34; - const points = chartPoints(observations, metric); - const columns = chartColumns(points); - const columnIndex = new Map(columns.map((column, index) => [column.key, index])); - - const seriesKeys = [...new Set(points.map((point) => point.seriesKey))]; - const height = legendTop + seriesKeys.length * 22 + 12; - const palette = ["#2563eb", "#7c3aed", "#0f766e", "#c2410c", "#be123c", "#4f46e5"]; - const color = new Map(seriesKeys.map((name, index) => [name, palette[index % palette.length]!])); - const seriesFieldsByKey = new Map( - seriesKeys.map((key) => [key, points.find((point) => point.seriesKey === key)!.fields]) - ); - - // Split shared context (rendered once as a subtitle) from the fields that - // distinguish the plotted series (rendered in each legend row). - const constantContext: string[] = []; - const varyingFields: Array<(typeof SERIES_CONTEXT_FIELDS)[number]> = []; - for (const field of SERIES_CONTEXT_FIELDS) { - const values = new Set([...seriesFieldsByKey.values()].map((fields) => fields[field])); - const sample = [...seriesFieldsByKey.values()][0]; - if (values.size <= 1) { - if (sample !== undefined) constantContext.push(sample[field]); - } else { - varyingFields.push(field); - } - } - const seriesLabel = (fields: SeriesFields): string => - [...varyingFields.map((field) => fields[field]), fields.target].join(" "); - - const available = points.map((point) => point.value).filter((value): value is number => value !== null); - const maxValue = ratioMetric ? 1 : Math.max(1, ...available); - const columnX = (column: ChartColumn): number => { - if (columns.length <= 1) return left + plotWidth / 2; - const index = columnIndex.get(column.key) ?? 0; - const pad = 44; - return left + pad + (index / (columns.length - 1)) * (plotWidth - 2 * pad); - }; - const x = (point: ChartPoint): number => { - const column = columns[columnIndex.get(chartColumnKey(point)) ?? 0]; - return column === undefined ? left + plotWidth / 2 : columnX(column); - }; - const y = (value: number): number => top + plotHeight - (value / maxValue) * plotHeight; - - const lines: string[] = [ - '', - ``, - `${xml(title)}`, - `${xml(`${title} by candidate commit and benchmark target${metric === "wall_clock_seconds" || metric === "cost_usd" ? "; partial values use hollow dashed markers, legacy partial values without a number use a dashed ring and partial n/a label, and unavailable values use an n/a cross" : ""}`)}`, - ``, - `${xml(title)}` - ]; - if (constantContext.length > 0) { - lines.push( - `${xml(constantContext.join(" · "))}` - ); - } - lines.push( - `${xml("Each line tracks one benchmark target across evenly spaced candidate-run columns.")}`, - ``, - `` - ); - for (let tick = 0; tick <= 4; tick += 1) { - const value = (maxValue * tick) / 4; - const tickY = y(value); - lines.push( - ``, - `${xml(formatMetric(value, ratioMetric))}` - ); - } - if (points.length === 0) { - lines.push( - `No published observations` - ); - } else { - for (const marker of chartLineageMarkers(points)) { - const column = columns[columnIndex.get(marker.columnKey) ?? 0]; - if (column === undefined) continue; - const markerX = columnX(column); - lines.push( - `${xml(marker.title)}`, - ``, - `${xml(marker.label)}` - ); - } - const bySeries = new Map(); - for (const point of points) bySeries.set(point.seriesKey, [...(bySeries.get(point.seriesKey) ?? []), point]); - for (const [name, values] of bySeries) { - const stroke = color.get(name)!; - const availableValues = values.filter((point): point is ChartPoint & { value: number } => point.value !== null); - if (availableValues.length > 1) { - lines.push( - `` - ); - } - for (const point of values) { - const pointX = x(point); - const commitUrl = `${point.repositoryUrl.replace(/\/$/u, "")}/commit/${point.commit}`; - const shortCommit = point.commit.slice(0, 7); - const label = seriesLabel(point.fields); - if (point.value === null) { - const pointY = plotBottom - 8; - const partialWithoutValue = point.completeness.status === "partial"; - const status = partialWithoutValue ? "partial (value unavailable)" : "unavailable"; - lines.push( - `${xml(`${label} ${shortCommit}: ${status}${point.completeness.reasons.length === 0 ? "" : ` (${point.completeness.reasons.join(", ")})`}`)}` - ); - if (partialWithoutValue) { - lines.push( - `` - ); - } - lines.push( - ``, - ``, - `${partialWithoutValue ? "partial n/a" : "n/a"} ${shortCommit}` - ); - continue; - } - const pointY = y(point.value); - const partial = point.completeness.status === "partial"; - lines.push( - `${xml(`${label} ${shortCommit}: ${formatMetric(point.value, ratioMetric)}${partial ? ` partial (${point.completeness.reasons.join(", ")})` : ""}`)}`, - partial - ? `` - : ``, - "" - ); - } - } - for (const column of columns) { - const labelX = columnX(column); - lines.push( - `${xml(column.commit.slice(0, 7))}`, - `${xml(column.timestamp.slice(0, 10))}` - ); - } - } - seriesKeys.forEach((name, index) => { - const fields = seriesFieldsByKey.get(name)!; - const rowY = legendTop + index * 22; - lines.push( - ``, - `${xml(seriesLabel(fields))}` - ); - }); - lines.push(""); - return `${lines.join("\n")}\n`; -} - function installHistoryPublication( historyPath: string, chartsDirectory: string, @@ -2835,10 +2430,6 @@ function formatCostTick(value: number): string { return `$${Number(value.toFixed(2)).toLocaleString("en-US", { useGrouping: false })}`; } -function formatMetric(value: number, ratioMetric: boolean): string { - return ratioMetric ? value.toFixed(2) : Number(value.toFixed(2)).toLocaleString("en-US", { useGrouping: false }); -} - function xml(value: string): string { return value.replace(/[&<>"']/gu, (character) => { switch (character) { diff --git a/packages/evals/src/index.ts b/packages/evals/src/index.ts index 2ac4ffeab..4ca6a4f67 100644 --- a/packages/evals/src/index.ts +++ b/packages/evals/src/index.ts @@ -9,13 +9,10 @@ export * from "./expansion.js"; export * from "./evaluator/adjudicator-prompt.js"; export * from "./evaluator/judge-panel.js"; export * from "./history.js"; -export * from "./history-publication.js"; export * from "./ground-truth.js"; -export * from "./node-telemetry.js"; export * from "./lineage.js"; export * from "./public-diagnostics.js"; export * from "./recovery-equivalence.js"; -export * from "./reporter.js"; export * from "./reporters/index.js"; export * from "./runner.js"; export * from "./scoring.js"; diff --git a/packages/evals/src/node-telemetry.ts b/packages/evals/src/node-telemetry.ts deleted file mode 100644 index 49f2fb394..000000000 --- a/packages/evals/src/node-telemetry.ts +++ /dev/null @@ -1,1008 +0,0 @@ -import fs from "node:fs"; -import path from "node:path"; -import { isDeepStrictEqual, TextDecoder } from "node:util"; - -import lockfile from "proper-lockfile"; - -import { - ARTIFACT_MANIFEST_FILE, - DEFAULT_STRICT_JSONL_MAX_BYTES, - DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES, - DEFAULT_STRICT_JSONL_MAX_RECORDS, - assertEventRecord, - assertNoSymlinkComponents, - assertRegularFileInside, - getNodeArtifactDir, - layoutForRunRoot, - normalizeSafeRelativePath, - parseStrictJson, - readArtifactManifest, - readRunState, - safeResolveInside, - sha256Bytes, - validateStrictJsonlHistory, - validateSafeId, - type ArtifactManifest, - type ArtifactManifestEntry, - type ArtifactManifestOutputContract, - type EventRecord, - type RunState -} from "@ultrafuzz/artifacts"; -import { - assertVerifiedFinalReportSnapshotRemainedCurrent, - loadVerifiedFinalReportSnapshot, - type RuntimeDiagnostic, - type VerifiedFinalReportSnapshot, - type VerifiedOutputArtifactSnapshot -} from "@ultrafuzz/runtime"; - -import { - reporterForReliableDelivery, - type EvalArtifactUpload, - type EvalNodeEvent, - type EvalNodeEventEnvelope, - type EvalReporter -} from "./reporter.js"; -import type { EvalMatrixRow, EvalReportingPolicy } from "./types.js"; -import { readTelemetryCursor, writeTelemetryCursor } from "./eval-durable.js"; -import { contentTypeForArtifact, EvalError, isRecord, warningDiagnostic } from "./utils.js"; -import { setTimeout as sleep } from "node:timers/promises"; - -export const TELEMETRY_CURSOR_SCHEMA_VERSION = "ultrafuzz.eval.telemetry-cursor.v1" as const; -const DELIVERED_EVENT_RING_SIZE = 4096; -const MAX_MANIFEST_BYTES = 1024 * 1024; -const SHA256_PATTERN = /^[a-f0-9]{64}$/u; -const TELEMETRY_CURSOR_LOCK_STALE_MS = 300_000; -const TELEMETRY_CURSOR_LOCK_FS = Object.assign(Object.create(fs) as typeof fs, { - stat: fs.lstat.bind(fs), - utimes: fs.lutimes.bind(fs) -}); - -export interface TelemetryCursorState { - schemaVersion: typeof TELEMETRY_CURSOR_SCHEMA_VERSION; - byteOffset: number; // position in events.jsonl - deliveredEventIds: string[]; // ring buffer, belt-and-braces dedup - uploadedArtifacts: Record; // `${nodeId}/${relativePath}` -> sha256 - lastHeartbeatAt: Record; // nodeId -> ISO, heartbeat rate limiting - providerIds: Record; // nodeId -> provider span/run id (rebuilt on resume) - /** findings-validated counts folded into node-finished events. */ - findingsCountByNode: Record; -} - -export interface TelemetryDrainResult { - deliveredEvents: number; - deliveredArtifacts: number; - warnings: RuntimeDiagnostic[]; -} - -export interface TelemetryDrainOptions { - /** Start this drain from a durably reset cursor while holding the cursor lease. */ - resetCursor?: boolean; -} - -export interface NodeTelemetryPumpInput { - /** Root of the underlying ultrafuzz run (contains events.jsonl, state.json, artifacts/). */ - runRoot: string; - row: EvalMatrixRow; - reporters: EvalReporter[]; - policy: EvalReportingPolicy; - cursorPath: string; - /** Exact preflight authority required by post-hoc publication. Live telemetry may omit it. */ - requiredFinalReportSnapshot?: VerifiedFinalReportSnapshot; - now?: () => Date; - maxDeliveryAttempts?: number; - retryDelayMs?: number; -} - -export function isRequiredFinalReportTelemetryError(error: unknown): error is EvalError { - return error instanceof EvalError && error.code.startsWith("EVAL_TELEMETRY_REQUIRED_FINAL_REPORT_"); -} - -export function createTelemetryCursor(): TelemetryCursorState { - return { - schemaVersion: TELEMETRY_CURSOR_SCHEMA_VERSION, - byteOffset: 0, - deliveredEventIds: [], - uploadedArtifacts: {}, - lastHeartbeatAt: {}, - providerIds: {}, - findingsCountByNode: {} - }; -} - -export function loadTelemetryCursor(cursorPath: string): TelemetryCursorState { - const absoluteCursorPath = path.resolve(cursorPath); - assertNoSymlinkComponents(path.parse(absoluteCursorPath).root, absoluteCursorPath, "telemetry cursor path"); - let stat: fs.Stats; - try { - stat = fs.lstatSync(cursorPath); - } catch (error) { - if (isErrnoException(error, "ENOENT")) { - return createTelemetryCursor(); - } - throw new EvalError( - "EVAL_TELEMETRY_CURSOR_READ_FAILED", - `failed to inspect telemetry cursor ${cursorPath}: ${error instanceof Error ? error.message : String(error)}`, - { path: cursorPath } - ); - } - if (stat.isSymbolicLink() || !stat.isFile()) { - throw new EvalError( - "EVAL_TELEMETRY_CURSOR_UNSAFE", - `telemetry cursor must be a regular file and cannot be a symbolic link: ${cursorPath}`, - { path: cursorPath } - ); - } - return readTelemetryCursor(cursorPath); -} - -/** - * Cursor over the run journal: reads new `events.jsonl` records after each - * sync tick, translates them into reporter envelopes, synthesizes heartbeats - * from `state.json`, and streams allowlisted artifacts from node manifests. - * - * Delivery is at-least-once. The cursor is persisted durably after callbacks, - * so a crash or cursor-write failure can replay a callback. Reporter-facing - * envelopes therefore carry stable idempotency keys. Delivery exhaustion - * degrades to warnings, while cursor lock/read/write failures are fatal. - */ -export class NodeTelemetryPump { - readonly cursor: TelemetryCursorState; - private readonly input: NodeTelemetryPumpInput; - private readonly now: () => Date; - private readonly maxAttempts: number; - private readonly retryDelayMs: number; - - constructor(input: NodeTelemetryPumpInput) { - this.input = input; - this.now = input.now ?? (() => new Date()); - this.maxAttempts = Math.max(1, input.maxDeliveryAttempts ?? 3); - this.retryDelayMs = input.retryDelayMs ?? 250; - this.cursor = createTelemetryCursor(); - } - - async drain(options: TelemetryDrainOptions = {}): Promise { - const release = await acquireTelemetryCursorLock(this.input.cursorPath); - try { - if (options.resetCursor === true) { - persistTelemetryCursor(this.input.cursorPath, createTelemetryCursor()); - } - const durableCursor = loadTelemetryCursor(this.input.cursorPath); - this.replaceCursor(durableCursor); - const rollbackCursor = cloneTelemetryCursor(durableCursor); - try { - return await this.drainLocked(); - } catch (error) { - this.replaceCursor(rollbackCursor); - throw error; - } - } finally { - await release(); - } - } - - private async drainLocked(): Promise { - const warnings: RuntimeDiagnostic[] = []; - const state = this.readState(); - const { records, nextOffset } = this.readNewJournalRecords(); - const replayOffset = this.cursor.byteOffset; - - const envelopes: EvalNodeEventEnvelope[] = []; - const uploads: EvalArtifactUpload[] = []; - for (const record of records) { - this.translate(record, state, envelopes, uploads, warnings); - } - envelopes.push(...this.synthesizeHeartbeats(state)); - - let deliveredEvents = 0; - let deliveryFailed = false; - for (const envelope of envelopes) { - if (this.cursor.deliveredEventIds.includes(envelope.eventId)) { - continue; - } - const delivered = await this.deliver( - `event ${envelope.event.type} (${envelope.nodeId})`, - (reporter) => reporter.onNodeEvent(envelope), - warnings - ); - if (!delivered) { - deliveryFailed = true; - continue; - } - this.markDelivered(envelope.eventId); - if (envelope.event.type === "node-heartbeat") { - this.cursor.lastHeartbeatAt[envelope.nodeId] = envelope.event.at; - } - deliveredEvents += 1; - } - - let deliveredArtifacts = 0; - for (const upload of uploads) { - const key = `${upload.nodeId}/${upload.relativePath}`; - if (this.cursor.uploadedArtifacts[key] === upload.sha256) { - continue; - } - const delivered = await this.deliver(`artifact ${key}`, (reporter) => reporter.onArtifact(upload), warnings); - if (!delivered) { - deliveryFailed = true; - continue; - } - this.cursor.uploadedArtifacts[key] = upload.sha256; - deliveredArtifacts += 1; - } - - // Re-read the journal after any failed callback. Successfully delivered - // IDs/hashes remain in the cursor, so resume retries only missing work. - this.cursor.byteOffset = deliveryFailed ? replayOffset : nextOffset; - this.persistCursor(); - return { deliveredEvents, deliveredArtifacts, warnings }; - } - - private readState(): RunState | undefined { - const statePath = path.join(this.input.runRoot, "state.json"); - const expectedRunId = layoutForRunRoot(this.input.runRoot).runId; - try { - fs.lstatSync(statePath); - } catch (error) { - if (isErrnoException(error, "ENOENT")) return undefined; - throw new EvalError( - "EVAL_TELEMETRY_STATE_READ_FAILED", - `failed to inspect run state ${statePath}: ${error instanceof Error ? error.message : String(error)}`, - { path: statePath } - ); - } - try { - assertRegularFileInside(this.input.runRoot, statePath, "telemetry run state"); - const state = readRunState(statePath); - if (state.run_id !== expectedRunId) { - throw new Error( - `run state belongs to ${JSON.stringify(state.run_id)}, expected ${JSON.stringify(expectedRunId)}` - ); - } - return state; - } catch (error) { - throw new EvalError( - "EVAL_TELEMETRY_STATE_READ_FAILED", - `failed to read run state ${statePath}: ${error instanceof Error ? error.message : String(error)}`, - { path: statePath } - ); - } - } - - private readNewJournalRecords(): { records: EventRecord[]; nextOffset: number } { - const eventsPath = path.join(this.input.runRoot, "events.jsonl"); - const expectedRunId = layoutForRunRoot(this.input.runRoot).runId; - let buffer: Buffer; - try { - try { - fs.lstatSync(eventsPath); - } catch (error) { - if (isErrnoException(error, "ENOENT")) { - if (this.cursor.byteOffset !== 0) { - throw new Error("event journal disappeared after the cursor advanced", { cause: error }); - } - return { records: [], nextOffset: this.cursor.byteOffset }; - } - throw error; - } - assertRegularFileInside(this.input.runRoot, eventsPath, "telemetry event journal"); - const flags = fs.constants.O_RDONLY | (fs.constants.O_NOFOLLOW ?? 0); - const fd = fs.openSync(eventsPath, flags); - let size: number; - try { - const stat = fs.fstatSync(fd); - if (!stat.isFile()) throw new Error("event journal is not a regular file"); - size = stat.size; - if (size < this.cursor.byteOffset) { - throw new Error("event journal is shorter than its durable cursor"); - } - if (size > DEFAULT_STRICT_JSONL_MAX_BYTES) { - throw new Error(`event journal exceeds the ${DEFAULT_STRICT_JSONL_MAX_BYTES}-byte limit`); - } - buffer = Buffer.alloc(size); - let bytesRead = 0; - while (bytesRead < buffer.byteLength) { - const count = fs.readSync(fd, buffer, bytesRead, buffer.byteLength - bytesRead, bytesRead); - if (count === 0) throw new Error("event journal changed while it was read"); - bytesRead += count; - } - } finally { - fs.closeSync(fd); - } - } catch (error) { - throw new EvalError( - "EVAL_TELEMETRY_JOURNAL_UNREADABLE", - `failed to read run journal ${eventsPath}: ${error instanceof Error ? error.message : String(error)}`, - { path: eventsPath } - ); - } - if (buffer.byteLength === 0) return { records: [], nextOffset: 0 }; - const lastNewline = buffer.lastIndexOf(0x0a); - if (lastNewline === -1) { - // Partial line only — wait for the writer to finish it. - return { records: [], nextOffset: this.cursor.byteOffset }; - } - const completeBytes = buffer.subarray(0, lastNewline + 1); - if (this.cursor.byteOffset > completeBytes.byteLength) { - throw new EvalError( - "EVAL_TELEMETRY_JOURNAL_UNREADABLE", - `run journal ${eventsPath} no longer has a complete record boundary at its durable cursor`, - { path: eventsPath } - ); - } - if (this.cursor.byteOffset > 0 && buffer[this.cursor.byteOffset - 1] !== 0x0a) { - throw new EvalError( - "EVAL_TELEMETRY_JOURNAL_UNREADABLE", - `run journal ${eventsPath} durable cursor is not at a record boundary`, - { path: eventsPath } - ); - } - let complete: string; - try { - complete = new TextDecoder("utf-8", { fatal: true }).decode(completeBytes); - } catch (error) { - throw new EvalError( - "EVAL_TELEMETRY_JOURNAL_MALFORMED", - `run journal ${eventsPath} contains invalid UTF-8; the cursor was not advanced`, - { path: eventsPath, reason: error instanceof Error ? error.message : String(error) } - ); - } - const nextOffset = completeBytes.byteLength; - const records: EventRecord[] = []; - const lines = complete.split("\n"); - lines.pop(); - if (lines.length > DEFAULT_STRICT_JSONL_MAX_RECORDS) { - throw new EvalError( - "EVAL_TELEMETRY_JOURNAL_MALFORMED", - `run journal ${eventsPath} exceeds the ${DEFAULT_STRICT_JSONL_MAX_RECORDS}-record limit`, - { path: eventsPath } - ); - } - for (const [index, line] of lines.entries()) { - try { - if (Buffer.byteLength(line, "utf8") > DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES) { - throw new Error(`record exceeds the ${DEFAULT_STRICT_JSONL_MAX_RECORD_BYTES}-byte limit`); - } - const record = assertEventRecord(parseStrictJson(line), `$[${index}]`); - if (record.run_id !== expectedRunId) { - throw new Error( - `record belongs to ${JSON.stringify(record.run_id)}, expected ${JSON.stringify(expectedRunId)}` - ); - } - records.push(record); - } catch (error) { - throw new EvalError( - "EVAL_TELEMETRY_JOURNAL_MALFORMED", - `run journal ${eventsPath} has an invalid record at line ${index + 1}; the cursor was not advanced: ${error instanceof Error ? error.message : String(error)}`, - { path: eventsPath } - ); - } - } - try { - validateStrictJsonlHistory(records, { - label: "telemetry event journal", - parseRecord: (value, recordPath) => assertEventRecord(value, recordPath), - identity: (record) => record.event_id, - validateHistory: (history) => { - let priorTimestamp: string | undefined; - for (const [index, record] of history.entries()) { - if (priorTimestamp !== undefined && record.timestamp < priorTimestamp) { - throw new Error(`event journal timestamps are not ordered at record ${index + 1}`); - } - priorTimestamp = record.timestamp; - } - } - }); - } catch (error) { - throw new EvalError( - "EVAL_TELEMETRY_JOURNAL_MALFORMED", - `run journal ${eventsPath} has invalid history; the cursor was not advanced: ${error instanceof Error ? error.message : String(error)}`, - { path: eventsPath } - ); - } - const priorRecordCount = countByte(buffer.subarray(0, this.cursor.byteOffset), 0x0a); - return { records: records.slice(priorRecordCount), nextOffset }; - } - - private translate( - record: EventRecord, - state: RunState | undefined, - envelopes: EvalNodeEventEnvelope[], - uploads: EvalArtifactUpload[], - warnings: RuntimeDiagnostic[] - ): void { - switch (record.event_type) { - case "node-synced": { - const nodeId = record.node_id; - const status = record.status; - if ( - status !== "running" && - status !== "succeeded" && - status !== "failed" && - status !== "timed-out" && - status !== "skipped" - ) { - return; - } - const attempt = requireTelemetryAttempt(record.payload.attempt, record.event_id); - if (status === "running") { - envelopes.push( - this.envelope(record.event_id, nodeId, { type: "node-started", at: record.timestamp, attempt }) - ); - return; - } - if (status === "succeeded" || status === "failed" || status === "timed-out" || status === "skipped") { - const nodeState = state?.nodes[nodeId]; - const event: EvalNodeEvent = { - type: "node-finished", - at: record.timestamp, - status, - attempt, - ...(nodeState?.started_at !== undefined ? { startedAt: nodeState.started_at } : {}), - ...(nodeState?.last_error !== undefined ? { error: nodeState.last_error } : {}), - ...(this.cursor.findingsCountByNode[nodeId] !== undefined - ? { findingsCount: this.cursor.findingsCountByNode[nodeId] } - : {}) - }; - envelopes.push(this.envelope(record.event_id, nodeId, event)); - } - return; - } - case "findings-validated": { - if (record.payload.count !== undefined) { - this.cursor.findingsCountByNode[record.node_id] = record.payload.count; - } - return; - } - case "artifact-manifest-written": { - const nodeId = record.node_id; - const manifest = this.readManifest(nodeId, warnings); - if (manifest === undefined) { - return; - } - envelopes.push( - this.envelope(record.event_id, nodeId, { - type: "node-artifacts", - at: record.timestamp, - manifest: manifest.files - }) - ); - uploads.push(...this.uploadsForManifest(nodeId, manifest, warnings)); - return; - } - default: - return; - } - } - - private readManifest(nodeId: string, warnings: RuntimeDiagnostic[]): ArtifactManifest | undefined { - let manifestPath: string; - let nodeDir: string; - let layout: ReturnType; - try { - layout = layoutForRunRoot(this.input.runRoot); - nodeDir = getNodeArtifactDir(layout, validateSafeId(nodeId, "node ID")); - manifestPath = safeResolveInside(nodeDir, ARTIFACT_MANIFEST_FILE, "artifact manifest path"); - } catch { - warnings.push(warningDiagnostic("EVAL_TELEMETRY_MANIFEST_UNSAFE", "skipped unsafe artifact manifest path")); - return undefined; - } - try { - fs.lstatSync(manifestPath); - } catch (error) { - if (isErrnoException(error, "ENOENT")) return undefined; - warnings.push( - warningDiagnostic( - "EVAL_TELEMETRY_MANIFEST_UNREADABLE", - `failed to inspect artifact manifest for ${nodeId}: ${error instanceof Error ? error.message : String(error)}` - ) - ); - return undefined; - } - try { - assertRegularFileInside(nodeDir, manifestPath, "artifact manifest path"); - if (fs.statSync(manifestPath).size > MAX_MANIFEST_BYTES) { - throw new Error("artifact manifest exceeds the size limit"); - } - const manifest = readArtifactManifest(layout, nodeId); - if (manifest.node_id !== nodeId || !Array.isArray(manifest.files)) { - throw new Error("artifact manifest does not match its node directory"); - } - for (const file of manifest.files as unknown[]) { - if (!isSafeManifestEntry(file)) { - throw new Error("artifact manifest contains an invalid file entry"); - } - } - return manifest; - } catch (error) { - warnings.push( - warningDiagnostic( - "EVAL_TELEMETRY_MANIFEST_UNREADABLE", - `artifact manifest for ${nodeId} is unreadable: ${error instanceof Error ? error.message : String(error)}` - ) - ); - return undefined; - } - } - - private uploadsForManifest( - nodeId: string, - manifest: ArtifactManifest, - warnings: RuntimeDiagnostic[] - ): EvalArtifactUpload[] { - const policy = this.input.policy.artifacts; - const includeSet = new Set(policy.include); - // Sensitivity gate: private targets stay manifest-only unless the suite - // explicitly opted into upload mode. - const payloadAllowed = - policy.mode === "upload" && (this.input.row.target.sensitivity !== "private" || policy.mode_explicit); - const uploads: EvalArtifactUpload[] = []; - const layout = layoutForRunRoot(this.input.runRoot); - const nodeDir = getNodeArtifactDir(layout, nodeId); - const requiredSnapshot = this.input.requiredFinalReportSnapshot; - const manifestDeclaresFinalReport = manifest.output_contracts.some( - (output) => output.contract === "ultrafuzz/report@3" - ); - const isRequiredFinalReportProducer = requiredSnapshot?.authority.attempt_id === nodeId; - const publishesFinalReport = manifestDeclaresFinalReport || isRequiredFinalReportProducer; - let verifiedFinalReport: Map | undefined; - if (publishesFinalReport) { - try { - const snapshot = requiredSnapshot ?? loadVerifiedFinalReportSnapshot(this.input.runRoot); - if (requiredSnapshot !== undefined) { - if (!isRequiredFinalReportProducer) { - throw new Error( - `manifest ${nodeId} declares a final report owned by ${requiredSnapshot.authority.attempt_id}` - ); - } - assertVerifiedFinalReportSnapshotRemainedCurrent(snapshot); - } - if (snapshot.authority.attempt_id !== nodeId) { - throw new Error(`verified final-report authority belongs to ${snapshot.authority.attempt_id}`); - } - verifiedFinalReport = verifiedFinalReportFiles(snapshot); - if (requiredSnapshot !== undefined) { - assertRequiredFinalReportManifest(manifest, snapshot, verifiedFinalReport); - } - } catch (error) { - if (requiredSnapshot !== undefined) { - throw new EvalError( - "EVAL_TELEMETRY_REQUIRED_FINAL_REPORT_AUTHORITY_INVALID", - `required final-report publication ${nodeId} lost its preflight authority: ${error instanceof Error ? error.message : String(error)}`, - { node_id: nodeId, reason: error instanceof Error ? error.message : String(error) } - ); - } - warnings.push( - warningDiagnostic( - "EVAL_TELEMETRY_ARTIFACT_UNVERIFIED", - `skipped unverified final-report publication ${nodeId}: ${error instanceof Error ? error.message : String(error)}` - ) - ); - return uploads; - } - } - for (const file of manifest.files) { - const verified = verifiedFinalReport?.get(file.path); - if (!includeSet.has(file.path) && (verified === undefined || !includeSet.has(verified.policyPath))) { - continue; - } - if (file.size_bytes > policy.max_file_bytes) { - continue; - } - if (publishesFinalReport && verified === undefined) continue; - if (verified !== undefined) { - if ( - verified.absolutePath !== safeResolveInside(nodeDir, file.path, "verified artifact upload path") || - verified.bytes.length !== file.size_bytes || - verified.sha256 !== file.sha256 - ) { - if (requiredSnapshot !== undefined) { - throw new EvalError( - "EVAL_TELEMETRY_REQUIRED_FINAL_REPORT_MANIFEST_MISMATCH", - `required final-report manifest differs from preflight bytes ${nodeId}/${file.path}`, - { node_id: nodeId, artifact_path: file.path } - ); - } - warnings.push( - warningDiagnostic( - "EVAL_TELEMETRY_ARTIFACT_UNVERIFIED", - `skipped final-report publication whose manifest differs from verified bytes ${nodeId}/${file.path}` - ) - ); - continue; - } - } else { - try { - validateArtifact(nodeDir, file, policy.max_file_bytes); - } catch (error) { - warnings.push( - warningDiagnostic( - "EVAL_TELEMETRY_ARTIFACT_UNSAFE", - `skipped unsafe artifact ${nodeId}/${file.path}: ${error instanceof Error ? error.message : String(error)}` - ) - ); - continue; - } - } - uploads.push({ - idempotencyKey: telemetryIdempotencyKey("artifact", [this.input.row.id, nodeId, file.path, file.sha256]), - rowId: this.input.row.id, - nodeId, - relativePath: file.path, - contentType: contentTypeForArtifact(file.path), - sizeBytes: file.size_bytes, - sha256: file.sha256, - ...(payloadAllowed - ? { - read: async () => { - if (verified === undefined) return readValidatedArtifact(nodeDir, file, policy.max_file_bytes); - if (requiredSnapshot !== undefined) { - try { - assertVerifiedFinalReportSnapshotRemainedCurrent(requiredSnapshot); - } catch (error) { - throw new EvalError( - "EVAL_TELEMETRY_REQUIRED_FINAL_REPORT_AUTHORITY_CHANGED", - `required final-report authority changed before delivering ${nodeId}/${file.path}`, - { - node_id: nodeId, - artifact_path: file.path, - reason: error instanceof Error ? error.message : String(error) - } - ); - } - } - return Buffer.from(verified.bytes); - } - } - : {}) - }); - } - return uploads; - } - - private synthesizeHeartbeats(state: RunState | undefined): EvalNodeEventEnvelope[] { - if (state === undefined) { - return []; - } - const intervalMs = this.input.policy.heartbeat_interval_seconds * 1000; - const now = this.now(); - const envelopes: EvalNodeEventEnvelope[] = []; - for (const [nodeId, node] of Object.entries(state.nodes)) { - if (node.status !== "running") { - continue; - } - const last = this.cursor.lastHeartbeatAt[nodeId]; - if (last !== undefined && now.getTime() - Date.parse(last) < intervalMs) { - continue; - } - const startedAt = node.started_at !== undefined ? Date.parse(node.started_at) : Number.NaN; - const activeSeconds = Number.isFinite(startedAt) - ? Math.max(0, Math.round((now.getTime() - startedAt) / 1000)) - : 0; - const bucket = Math.floor(now.getTime() / intervalMs); - envelopes.push( - this.envelope(`evt-hb-${nodeId}-${bucket}`, nodeId, { - type: "node-heartbeat", - at: now.toISOString(), - status: node.retry_count > 0 ? "retrying" : "running", - activeSeconds - }) - ); - } - return envelopes; - } - - private envelope(eventId: string, nodeId: string, event: EvalNodeEvent): EvalNodeEventEnvelope { - return { - eventId, - idempotencyKey: telemetryIdempotencyKey("event", [this.input.row.id, eventId]), - rowId: this.input.row.id, - nodeId, - event - }; - } - - private async deliver( - label: string, - action: (reporter: EvalReporter) => Promise, - warnings: RuntimeDiagnostic[] - ): Promise { - let allDelivered = true; - for (const configuredReporter of this.input.reporters) { - const reporter = reporterForReliableDelivery(configuredReporter); - let lastError: unknown; - let delivered = false; - for (let attempt = 1; attempt <= this.maxAttempts && !delivered; attempt += 1) { - try { - await action(reporter); - delivered = true; - } catch (error) { - if (isRequiredFinalReportTelemetryError(error)) throw error; - lastError = error; - if (attempt < this.maxAttempts && this.retryDelayMs > 0) { - await sleep(this.retryDelayMs * attempt); - } - } - } - if (!delivered) { - allDelivered = false; - warnings.push( - warningDiagnostic( - "EVAL_TELEMETRY_DELIVERY_FAILED", - `${reporter.name} failed to deliver ${label}: ${ - lastError instanceof Error ? lastError.message : String(lastError) - }` - ) - ); - } - } - return allDelivered; - } - - private markDelivered(eventId: string): void { - this.cursor.deliveredEventIds.push(eventId); - if (this.cursor.deliveredEventIds.length > DELIVERED_EVENT_RING_SIZE) { - this.cursor.deliveredEventIds.splice(0, this.cursor.deliveredEventIds.length - DELIVERED_EVENT_RING_SIZE); - } - } - - private persistCursor(): void { - persistTelemetryCursor(this.input.cursorPath, this.cursor); - } - - private replaceCursor(next: TelemetryCursorState): void { - const clone = cloneTelemetryCursor(next); - this.cursor.schemaVersion = clone.schemaVersion; - this.cursor.byteOffset = clone.byteOffset; - this.cursor.deliveredEventIds = clone.deliveredEventIds; - this.cursor.uploadedArtifacts = clone.uploadedArtifacts; - this.cursor.lastHeartbeatAt = clone.lastHeartbeatAt; - this.cursor.providerIds = clone.providerIds; - this.cursor.findingsCountByNode = clone.findingsCountByNode; - } -} - -interface VerifiedFinalReportFile { - bytes: Buffer; - sha256: string; - absolutePath: string; - /** Stable reporting-policy role retained when the topology uses a custom path. */ - policyPath: "report.json" | "report.md"; -} - -function verifiedFinalReportFiles(snapshot: VerifiedFinalReportSnapshot): Map { - const bindings = [ - { - contract: "ultrafuzz/report@3", - policyPath: "report.json", - absolutePath: snapshot.artifacts.json_path, - bytes: snapshot.json_bytes - }, - { - contract: "ultrafuzz/nonempty-markdown@1", - policyPath: "report.md", - absolutePath: snapshot.artifacts.markdown_path, - bytes: snapshot.markdown_bytes - } - ] as const; - const files = new Map(); - for (const binding of bindings) { - const matches = snapshot.authority.outputs.filter( - (output) => - output.contract === binding.contract && - path.resolve(output.absolute_path) === path.resolve(binding.absolutePath) - ); - if (matches.length !== 1) { - throw new Error(`verified final-report snapshot does not bind one exact ${binding.contract} declaration`); - } - const output = matches[0]!; - if (files.has(output.path)) { - throw new Error(`verified final-report snapshot repeats declared path ${output.path}`); - } - const bytes = Buffer.from(binding.bytes); - files.set(output.path, { - bytes, - sha256: sha256Bytes(bytes), - absolutePath: binding.absolutePath, - policyPath: binding.policyPath - }); - } - return files; -} - -function assertRequiredFinalReportManifest( - manifest: ArtifactManifest, - snapshot: VerifiedFinalReportSnapshot, - verifiedFiles: ReadonlyMap -): void { - const attemptId = snapshot.authority.attempt_id; - const expectedRunId = layoutForRunRoot(snapshot.authority.run_root).runId; - if (manifest.run_id !== expectedRunId || manifest.node_id !== attemptId || manifest.producer_node_id !== attemptId) { - throw new Error(`live final-report manifest identity does not match required producer ${attemptId}`); - } - - const expectedDeclarations = finalReportDeclarations(snapshot.authority.outputs); - const actualDeclarations = finalReportDeclarations(manifest.output_contracts); - if (expectedDeclarations.length !== 2 || !isDeepStrictEqual(actualDeclarations, expectedDeclarations)) { - throw new Error("live final-report manifest does not declare the exact required JSON/Markdown pair"); - } - - for (const [relativePath, verified] of verifiedFiles) { - const entries = manifest.files.filter((file) => file.path === relativePath); - if (entries.length !== 1) { - throw new Error(`live final-report manifest must contain exactly one file entry for ${relativePath}`); - } - const entry = entries[0]!; - if (entry.size_bytes !== verified.bytes.length || entry.sha256 !== verified.sha256) { - throw new Error(`live final-report manifest file binding changed for ${relativePath}`); - } - } -} - -function finalReportDeclarations( - outputs: readonly (ArtifactManifestOutputContract | VerifiedOutputArtifactSnapshot)[] -): ArtifactManifestOutputContract[] { - return outputs - .filter((output) => output.contract === "ultrafuzz/report@3" || output.contract === "ultrafuzz/nonempty-markdown@1") - .map((output) => ({ - path: output.path, - contract: output.contract, - contract_digest: output.contract_digest, - ...(output.schema_file === undefined ? {} : { schema_file: output.schema_file }), - ...(output.schema_id === undefined ? {} : { schema_id: output.schema_id }), - ...(output.schema_sha256 === undefined ? {} : { schema_sha256: output.schema_sha256 }), - ...(output.schema_bundle_sha256 === undefined ? {} : { schema_bundle_sha256: output.schema_bundle_sha256 }), - ...(output.validator_build === undefined ? {} : { validator_build: output.validator_build }), - primary: output.primary - })) - .sort((left, right) => left.contract.localeCompare(right.contract) || left.path.localeCompare(right.path)); -} - -async function acquireTelemetryCursorLock(cursorPath: string): Promise<() => Promise> { - const absoluteCursorPath = path.resolve(cursorPath); - const directory = path.dirname(absoluteCursorPath); - const filesystemRoot = path.parse(absoluteCursorPath).root; - assertNoSymlinkComponents(filesystemRoot, directory, "telemetry cursor directory"); - fs.mkdirSync(directory, { recursive: true }); - assertPhysicalTelemetryCursorDirectory(absoluteCursorPath); - const lockPath = `${absoluteCursorPath}.lock`; - try { - const lockStat = fs.lstatSync(lockPath); - if (lockStat.isSymbolicLink()) { - throw new EvalError( - "EVAL_TELEMETRY_CURSOR_UNSAFE", - `telemetry cursor lock cannot be a symbolic link: ${lockPath}`, - { path: lockPath } - ); - } - } catch (error) { - if (!isErrnoException(error, "ENOENT")) { - throw error; - } - } - try { - return await lockfile.lock(absoluteCursorPath, { - lockfilePath: lockPath, - realpath: false, - fs: TELEMETRY_CURSOR_LOCK_FS, - stale: TELEMETRY_CURSOR_LOCK_STALE_MS, - update: 60_000, - retries: { retries: 120, factor: 1, minTimeout: 25, maxTimeout: 250 } - }); - } catch (error) { - throw new EvalError( - "EVAL_TELEMETRY_CURSOR_LOCK_FAILED", - `failed to acquire telemetry cursor lock ${lockPath}: ${error instanceof Error ? error.message : String(error)}`, - { path: lockPath } - ); - } -} - -function cloneTelemetryCursor(cursor: TelemetryCursorState): TelemetryCursorState { - return { - schemaVersion: cursor.schemaVersion, - byteOffset: cursor.byteOffset, - deliveredEventIds: [...cursor.deliveredEventIds], - uploadedArtifacts: { ...cursor.uploadedArtifacts }, - lastHeartbeatAt: { ...cursor.lastHeartbeatAt }, - providerIds: { ...cursor.providerIds }, - findingsCountByNode: { ...cursor.findingsCountByNode } - }; -} - -function persistTelemetryCursor(cursorPath: string, cursor: TelemetryCursorState): void { - try { - assertPhysicalTelemetryCursorDirectory(cursorPath); - writeTelemetryCursor(cursorPath, cursor); - } catch (error) { - throw new EvalError( - "EVAL_TELEMETRY_CURSOR_WRITE_FAILED", - `failed to persist telemetry cursor ${cursorPath}: ${error instanceof Error ? error.message : String(error)}`, - { path: cursorPath } - ); - } -} - -function assertPhysicalTelemetryCursorDirectory(cursorPath: string): void { - const absoluteCursorPath = path.resolve(cursorPath); - const directory = path.dirname(absoluteCursorPath); - assertNoSymlinkComponents(path.parse(absoluteCursorPath).root, directory, "telemetry cursor directory"); - const directoryStat = fs.lstatSync(directory); - if (directoryStat.isSymbolicLink() || !directoryStat.isDirectory()) { - throw new EvalError( - "EVAL_TELEMETRY_CURSOR_UNSAFE", - `telemetry cursor directory must be a physical directory: ${directory}`, - { path: directory } - ); - } -} - -function telemetryIdempotencyKey(kind: "event" | "artifact", components: readonly string[]): string { - const digest = sha256Bytes(Buffer.from(JSON.stringify(components), "utf8")); - return `ultrafuzz-${kind}-${digest}`; -} - -function isErrnoException(error: unknown, code: string): error is NodeJS.ErrnoException { - return error instanceof Error && "code" in error && (error as NodeJS.ErrnoException).code === code; -} - -function isSafeManifestEntry(value: unknown): value is ArtifactManifestEntry { - if ( - !isRecord(value) || - typeof value.path !== "string" || - !Number.isSafeInteger(value.size_bytes) || - (value.size_bytes as number) < 0 || - typeof value.sha256 !== "string" || - !SHA256_PATTERN.test(value.sha256) || - !isRecord(value.provenance) - ) { - return false; - } - try { - return normalizeSafeRelativePath(value.path, "artifact manifest file path") === value.path; - } catch { - return false; - } -} - -function validateArtifact(nodeDir: string, file: ArtifactManifestEntry, maxFileBytes: number): void { - void readValidatedArtifact(nodeDir, file, maxFileBytes); -} - -function readValidatedArtifact(nodeDir: string, file: ArtifactManifestEntry, maxFileBytes: number): Buffer { - const absolutePath = safeResolveInside(nodeDir, file.path, "artifact upload path"); - assertRegularFileInside(nodeDir, absolutePath, "artifact upload path"); - const noFollow = (fs.constants as typeof fs.constants & { O_NOFOLLOW?: number }).O_NOFOLLOW ?? 0; - const descriptor = fs.openSync(absolutePath, fs.constants.O_RDONLY | noFollow); - try { - const stat = fs.fstatSync(descriptor); - if (!stat.isFile() || stat.size > maxFileBytes || stat.size !== file.size_bytes) { - throw new Error("artifact size does not match its manifest or exceeds the upload limit"); - } - const contents = fs.readFileSync(descriptor); - if (sha256Bytes(contents) !== file.sha256) { - throw new Error("artifact digest does not match its manifest"); - } - return contents; - } finally { - fs.closeSync(descriptor); - } -} - -function requireTelemetryAttempt(attempt: number | undefined, eventId: string): number { - if (attempt === undefined || attempt < 1) { - throw new EvalError( - "EVAL_TELEMETRY_EVENT_UNUSABLE", - `node transition ${eventId} must carry a positive canonical payload.attempt` - ); - } - return attempt; -} - -function countByte(bytes: Buffer, expected: number): number { - let count = 0; - for (const byte of bytes) { - if (byte === expected) count += 1; - } - return count; -} diff --git a/packages/evals/src/report-authority.ts b/packages/evals/src/report-authority.ts index 5bd3588ed..47786287e 100644 --- a/packages/evals/src/report-authority.ts +++ b/packages/evals/src/report-authority.ts @@ -1,5 +1,4 @@ import path from "node:path"; -import { isDeepStrictEqual } from "node:util"; import { readRunState, sha256Bytes, type RunState } from "@ultrafuzz/artifacts"; import { @@ -103,21 +102,6 @@ export function loadBoundEvalReportAuthority( }; } -/** Require a persisted score authority to remain the exact current report authority. */ -export function assertEvalReportAuthorityRemainedCurrent( - record: EvalRunRecord, - expected: EvalReportAuthority -): BoundEvalReportAuthority { - const current = loadBoundEvalReportAuthority(record); - if (!isDeepStrictEqual(current.authority, expected)) { - throw invalidReportAuthority(record, "persisted score authority does not match the current verified report", { - expected, - current: current.authority - }); - } - return current; -} - function exactSnapshotOutput( snapshot: VerifiedFinalReportSnapshot, artifactPath: string, diff --git a/packages/evals/src/reporter.ts b/packages/evals/src/reporter.ts deleted file mode 100644 index e8a6ec3d0..000000000 --- a/packages/evals/src/reporter.ts +++ /dev/null @@ -1,213 +0,0 @@ -import { assertPlannedGraph, type ArtifactManifestEntry } from "@ultrafuzz/artifacts"; -import type { RuntimeDiagnostic } from "@ultrafuzz/runtime"; - -import type { EvalMatrixRow, EvalPlanValue, EvalRecoveryEquivalence, EvalRowScore, EvalScoreSummary } from "./types.js"; -import { warningDiagnostic } from "./utils.js"; - -/** The plan handed to reporters is the local plan value — providers never shape it. */ -export type EvalPlan = EvalPlanValue; - -/** Local score summary mirrored out to providers; grading never depends on them. */ -export type EvalSummary = EvalScoreSummary; - -/** Terminal outcome of one matrix row (one full ultrafuzz run). */ -export interface EvalRowResult { - status: "succeeded" | "failed" | "timed-out" | "canceled" | "launched"; - runId?: string; - runRoot?: string; - startedAt?: string; - finishedAt?: string; - graphFingerprint?: string; - configFingerprint?: string; - executionArtifactId?: string; - recoveryEquivalence?: EvalRecoveryEquivalence; - diagnostics?: RuntimeDiagnostic[]; -} - -/** Topology handed to the provider once per row, derived from the run's graph.json. */ -export interface EvalRowGraph { - rowId: string; - nodes: Array<{ - id: string; // concrete node id (fanout-expanded attempt) - logicalId: string; // node id as written in topology.yml - kind: "agentic" | "meta" | "reference"; - group: string; // setup | properties | strategies | references | review - dependsOn: string[]; // DAG edges from topology.yml - modelProfileId?: string; - model?: string; - loopIndex?: number; - }>; -} - -export type EvalNodeEvent = - | { type: "node-started"; at: string; attempt: number } - | { - type: "node-heartbeat"; - at: string; - status: "running" | "retrying" | "waiting-approval"; - activeSeconds: number; - } - | { - type: "node-finished"; - at: string; - status: "succeeded" | "failed" | "timed-out" | "skipped"; - startedAt?: string; - attempt: number; - error?: string; - findingsCount?: number; - } - | { type: "node-artifacts"; at: string; manifest: ArtifactManifestEntry[] }; - -export interface EvalNodeEventEnvelope { - eventId: string; // source events.jsonl event_id - /** Stable, row-scoped key reporters must use to make at-least-once delivery idempotent. */ - idempotencyKey: string; - rowId: string; - nodeId: string; // joins to EvalRowGraph.nodes[].id - event: EvalNodeEvent; -} - -export interface EvalArtifactUpload { - /** Stable key over row, node, relative path, and content digest. */ - idempotencyKey: string; - rowId: string; - nodeId: string; - relativePath: string; // e.g. "report.md", "report.json" - contentType: string; - sizeBytes: number; - sha256: string; - /** Absent in manifest-only mode (private targets): publish metadata, not payload. */ - read?: () => Promise; -} - -/** - * Providers are pure observers/exporters. Ultrafuzz owns the loop - * (plan → run → score → summarize, all writing local artifacts); a reporter - * observes that stream through explicitly supplied callbacks. No built-in - * remote reporter or configuration-based exporter is installed. - */ -export interface EvalReporter { - readonly name: string; - onPlan(plan: EvalPlan): Promise; - onRowStart(row: EvalMatrixRow, graph: EvalRowGraph): Promise; - onNodeEvent(envelope: EvalNodeEventEnvelope): Promise; - onArtifact(artifact: EvalArtifactUpload): Promise; - onRowFinish(row: EvalMatrixRow, result: EvalRowResult): Promise; - onScores(scores: EvalRowScore[], summary: EvalSummary): Promise; - finalize(summary: EvalSummary): Promise<{ url?: string }>; -} - -const KNOWN_GROUPS = ["setup", "properties", "strategies", "references", "review"]; -export const DEFAULT_GRAPH_GROUP = "default"; - -const guardedReporterTargets = new WeakMap(); - -/** - * Build the provider-facing row graph from a run's graph.json (PlannedGraph). - * An absent graph is allowed while a detached run is still planning. Once a - * graph is present, telemetry observes the canonical contract without repair. - */ -export function graphFromPlannedGraph(graph: unknown, rowId: string): EvalRowGraph { - if (graph === undefined) { - return { rowId, nodes: [] }; - } - const planned = assertPlannedGraph(graph); - const groupNames = Object.keys(planned.groups); - return { - rowId, - nodes: planned.nodes.map((node) => { - const fanout = node.model_fanout[0]; - return { - id: node.id, - logicalId: node.logical_id, - kind: node.kind, - group: resolveGroup(node.logical_id, groupNames), - dependsOn: [...node.depends_on], - ...(fanout === undefined ? {} : { modelProfileId: fanout.model_profile_id }), - ...(fanout?.model_name === undefined ? {} : { model: fanout.model_name }), - loopIndex: node.loop.index - }; - }) - }; -} - -export function groupsInGraph(graph: EvalRowGraph): string[] { - return [...new Set(graph.nodes.map((node) => node.group))].sort(); -} - -function resolveGroup(logicalId: string, groupNames: string[]): string { - for (const names of [groupNames, KNOWN_GROUPS]) { - for (const name of names) { - if (logicalId === name || logicalId.startsWith(`${name}-`) || logicalId.startsWith(`${name}.`)) { - return name; - } - } - } - return DEFAULT_GRAPH_GROUP; -} - -/** - * Wrap a reporter so that every callback failure degrades to a warning - * diagnostic instead of an exception — a provider outage must never kill an - * eval run. - */ -export function guardReporter( - reporter: EvalReporter, - onWarning: (diagnostic: RuntimeDiagnostic) => void -): EvalReporter { - const guard = async (operation: string, action: () => Promise): Promise => { - try { - await action(); - } catch (error) { - onWarning( - warningDiagnostic( - "EVAL_REPORTER_CALLBACK_FAILED", - `${reporter.name} ${operation} failed: ${error instanceof Error ? error.message : String(error)}` - ) - ); - } - }; - const guarded: EvalReporter = { - name: reporter.name, - onPlan: (plan) => guard("onPlan", () => reporter.onPlan(plan)), - onRowStart: (row, graph) => guard("onRowStart", () => reporter.onRowStart(row, graph)), - onNodeEvent: (envelope) => guard("onNodeEvent", () => reporter.onNodeEvent(envelope)), - onArtifact: (artifact) => guard("onArtifact", () => reporter.onArtifact(artifact)), - onRowFinish: (row, result) => guard("onRowFinish", () => reporter.onRowFinish(row, result)), - onScores: (scores, summary) => guard("onScores", () => reporter.onScores(scores, summary)), - finalize: async (summary) => { - try { - return await reporter.finalize(summary); - } catch (error) { - onWarning( - warningDiagnostic( - "EVAL_REPORTER_CALLBACK_FAILED", - `${reporter.name} finalize failed: ${error instanceof Error ? error.message : String(error)}` - ) - ); - return {}; - } - } - }; - guardedReporterTargets.set(guarded, reporter); - return guarded; -} - -/** - * Return the underlying reporter for delivery loops that own their retry and - * warning semantics. Other call sites keep using the guarded facade so a - * provider outage cannot abort an eval run. - */ -export function reporterForReliableDelivery(reporter: EvalReporter): EvalReporter { - let current = reporter; - const seen = new Set(); - while (!seen.has(current)) { - seen.add(current); - const target = guardedReporterTargets.get(current); - if (target === undefined) { - return current; - } - current = target; - } - return current; -} diff --git a/packages/evals/src/reporters/http.ts b/packages/evals/src/reporters/http.ts deleted file mode 100644 index 138ba0b9f..000000000 --- a/packages/evals/src/reporters/http.ts +++ /dev/null @@ -1,65 +0,0 @@ -import { EvalError } from "../utils.js"; - -const MAX_PROVIDER_RESPONSE_BYTES = 1024 * 1024; - -export async function boundedResponseText( - response: Response, - label: string, - errorCode: string, - maxBytes = MAX_PROVIDER_RESPONSE_BYTES -): Promise { - return (await boundedResponseBytes(response, label, errorCode, maxBytes)).toString("utf8"); -} - -export async function boundedResponseBytes( - response: Response, - label: string, - errorCode: string, - maxBytes = MAX_PROVIDER_RESPONSE_BYTES -): Promise { - const declaredLength = response.headers?.get?.("content-length"); - if (declaredLength !== null && declaredLength !== undefined) { - const declaredBytes = Number(declaredLength); - if (Number.isFinite(declaredBytes) && declaredBytes > maxBytes) { - await response.body?.cancel(); - throw responseTooLarge(label, errorCode, maxBytes); - } - } - - if (response.body === null || response.body === undefined || typeof response.body.getReader !== "function") { - const text = await response.text(); - const bytes = Buffer.from(text, "utf8"); - if (bytes.byteLength > maxBytes) { - throw responseTooLarge(label, errorCode, maxBytes); - } - return bytes; - } - - const chunks: Uint8Array[] = []; - let totalBytes = 0; - const reader = response.body.getReader(); - try { - while (true) { - const { done, value } = await reader.read(); - if (done) { - break; - } - totalBytes += value.byteLength; - if (totalBytes > maxBytes) { - await reader.cancel(); - throw responseTooLarge(label, errorCode, maxBytes); - } - chunks.push(value); - } - } finally { - reader.releaseLock(); - } - return Buffer.concat( - chunks.map((chunk) => Buffer.from(chunk)), - totalBytes - ); -} - -function responseTooLarge(label: string, code: string, maxBytes: number): EvalError { - return new EvalError(code, `${label} response exceeded ${maxBytes} bytes`, { label, maxBytes }); -} diff --git a/packages/evals/src/runner.ts b/packages/evals/src/runner.ts index 66a2e5d55..a2edac233 100644 --- a/packages/evals/src/runner.ts +++ b/packages/evals/src/runner.ts @@ -1,13 +1,7 @@ import fs from "node:fs"; import path from "node:path"; -import { - readPlannedGraphDocument, - readRunPlanDocument, - readRunState, - writeFileDurable, - type RunState -} from "@ultrafuzz/artifacts"; +import { readRunPlanDocument, readRunState, writeFileDurable, type RunState } from "@ultrafuzz/artifacts"; import { auditProfile, loadAuditProfileCatalog, @@ -28,14 +22,12 @@ import { BENCHMARK_SMOKE_WORKFLOW_PROFILE } from "./benchmark-manifest.js"; import { evalWorkflowLifecycle, isTerminalWorkflowStatus } from "./efficiency.js"; import { appendEvalRunRecord, writeEvalMatrix, writeEvalRunManifest, writeEvalRunSummary } from "./eval-durable.js"; import { evalRunExpansion } from "./expansion.js"; -import { NodeTelemetryPump } from "./node-telemetry.js"; import { buildEvalRunProvenance, DEFAULT_EVAL_POLL_INTERVAL_MS, DEFAULT_EVAL_WATCH_TIMEOUT_SECONDS } from "./lineage.js"; import { classifyRecoveryEquivalence } from "./recovery-equivalence.js"; -import { graphFromPlannedGraph, type EvalReporter, type EvalRowResult } from "./reporter.js"; import { resolveEvalProvider } from "./reporters/index.js"; import { evalWorkflowInputSchema, planEvalSuite, type PlanEvalSuiteInput } from "./suite.js"; import { @@ -103,7 +95,7 @@ export interface RunEvalSuiteInput extends PlanEvalSuiteInput { provider?: string; /** `[eval]` section of resolved config; only the final provider selection is inspected. */ evalProviderConfig?: EvalConfig; - /** Poll runs to terminal state and stream node telemetry (default: reporting.node_telemetry). */ + /** Poll runs to terminal state (default: reporting.node_telemetry). */ watch?: boolean; watchTimeoutSeconds?: number; pollIntervalMs?: number; @@ -142,7 +134,6 @@ export async function runEvalSuite(input: RunEvalSuiteInput): Promise { @@ -189,7 +177,6 @@ export async function runEvalSuite(input: RunEvalSuiteInput): Promise; sync?: RowSync; @@ -585,9 +571,9 @@ export interface WatchEvalRowInput { } /** - * Drive the row's telemetry pump from the driver poll loop: sync the detached - * workflow, drain the journal after every tick, and finish with one final - * catch-up drain plus `onRowFinish` once the run reaches a terminal state. + * Poll the detached workflow until its durable state is terminal or the watch + * deadline passes: sync, read `state.json`, sleep. A failed sync is counted and + * recorded on the row; it does not end the watch. */ export async function watchEvalRow( input: WatchEvalRowInput @@ -597,13 +583,6 @@ export async function watchEvalRow( if (runRoot === undefined) { return { record: input.record, diagnostics }; } - const pump = new NodeTelemetryPump({ - runRoot, - row: input.row, - reporters: input.reporters, - policy: input.plan.suite.reporting, - cursorPath: path.join(input.evalRunRoot, "telemetry", `${input.row.id}.cursor.json`) - }); const sync: RowSync = input.sync ?? defaultRowSync; const pollIntervalMs = input.pollIntervalMs ?? DEFAULT_EVAL_POLL_INTERVAL_MS; const deadline = Date.now() + (input.timeoutSeconds ?? DEFAULT_EVAL_WATCH_TIMEOUT_SECONDS) * 1000; @@ -613,27 +592,6 @@ export async function watchEvalRow( let lastSyncFailureAt: string | undefined; let lastSyncFailureMessage: string | undefined; - // The detached subprocess writes graph.json only after DAG planning, which - // can be seconds to tens of seconds after launch. Defer onRowStart until the - // graph is on disk so reporters see the real node list (an empty graph would - // orphan every node run under a never-created "default" group). `force` - // falls back to the empty graph so onRowStart always precedes drains/finish. - let rowStarted = false; - const startRowIfReady = async (force: boolean): Promise => { - if (rowStarted) { - return; - } - const graph = readGraph(runRoot); - if (graph === undefined && !force) { - return; - } - rowStarted = true; - for (const reporter of input.reporters) { - await reporter.onRowStart(input.row, graphFromPlannedGraph(graph, input.row.id)); - } - }; - - await startRowIfReady(false); let state = readStateSafe(runRoot, input.record.ultrafuzz_run_id); while (state !== undefined && !isTerminalRunStatus(state.status) && Date.now() < deadline) { try { @@ -652,11 +610,6 @@ export async function watchEvalRow( lastSyncFailureAt = observedAt; lastSyncFailureMessage = error instanceof Error ? error.message : String(error); } - await startRowIfReady(false); - if (rowStarted) { - const drained = await pump.drain(); - diagnostics.push(...drained.warnings); - } state = readStateSafe(runRoot, input.record.ultrafuzz_run_id); if (state !== undefined && isTerminalRunStatus(state.status)) { break; @@ -664,10 +617,8 @@ export async function watchEvalRow( await sleep(pollIntervalMs); } - // Final catch-up after the row reaches terminal state (or times out). - await startRowIfReady(true); - const finalDrain = await pump.drain(); - diagnostics.push(...finalDrain.warnings); + // Other syncers (`ultrafuzz status`, the dashboard) also write state.json; a + // run they finished during the final sleep is terminal, not timed out. state = readStateSafe(runRoot, input.record.ultrafuzz_run_id); const watchTimedOut = Date.now() >= deadline && (state === undefined || !isTerminalRunStatus(state.status)); const syncFailureDiagnostic: RuntimeDiagnostic | undefined = @@ -702,26 +653,9 @@ export async function watchEvalRow( policy: input.plan.suite.recovery_equivalence }) : undefined; - const result: EvalRowResult = { - status: watchTimedOut ? "timed-out" : rowStatus(state), - ...(input.record.ultrafuzz_run_id !== undefined ? { runId: input.record.ultrafuzz_run_id } : {}), - runRoot, - ...(state?.started_at !== undefined ? { startedAt: state.started_at } : {}), - ...(state?.finished_at !== undefined ? { finishedAt: state.finished_at } : {}), - ...(input.record.graph_fingerprint !== undefined ? { graphFingerprint: input.record.graph_fingerprint } : {}), - ...(input.record.config_fingerprint !== undefined ? { configFingerprint: input.record.config_fingerprint } : {}), - ...(input.record.execution_artifact_id !== undefined - ? { executionArtifactId: input.record.execution_artifact_id } - : {}), - ...(recoveryEquivalence === undefined ? {} : { recoveryEquivalence }), - diagnostics - }; - for (const reporter of input.reporters) { - await reporter.onRowFinish(input.row, result); - } const updatedRecord: EvalRunRecord = { ...input.record, - final_status: result.status, + final_status: watchTimedOut ? "timed-out" : rowStatus(state), ...(state === undefined ? {} : { @@ -761,19 +695,6 @@ const defaultRowSync: RowSync = async (input) => { } }; -function readGraph(runRoot: string): unknown { - const graphPath = path.join(runRoot, "graph.json"); - try { - fs.lstatSync(graphPath); - } catch (error) { - if (error instanceof Error && "code" in error && error.code === "ENOENT") { - return undefined; - } - throw error; - } - return readPlannedGraphDocument(graphPath); -} - function readStateSafe(runRoot: string, expectedRunId?: string): RunState | undefined { const statePath = path.join(runRoot, "state.json"); try { @@ -880,7 +801,7 @@ export function isTerminalRunStatus(status: string): boolean { return isTerminalWorkflowStatus(status); } -function rowStatus(state: RunState | undefined): EvalRowResult["status"] { +function rowStatus(state: RunState | undefined): NonNullable { if (state === undefined) { return "launched"; } diff --git a/packages/evals/src/scoring.ts b/packages/evals/src/scoring.ts index 29687da60..d54cb3240 100644 --- a/packages/evals/src/scoring.ts +++ b/packages/evals/src/scoring.ts @@ -33,7 +33,6 @@ import { validateEvalJsonSchema, type EVAL_LLM_JUDGE_RESULT_SCHEMA_VERSION } from "./eval-schema-registry.js"; -import { boundedResponseText } from "./reporters/http.js"; import { buildEvalSummaryProvenance } from "./lineage.js"; import { assertGroundTruthSubject, @@ -1076,7 +1075,7 @@ async function requestJudgeOnce(request: JudgeGatewayRequest, input: FindingJudg body: request.body, signal: controller.signal }); - const bodyText = await boundedResponseText(response, "LLM judge", "EVAL_LLM_JUDGE_RESPONSE_TOO_LARGE"); + const bodyText = await boundedJudgeResponseText(response); if (!response.ok) { throw new EvalError("EVAL_LLM_JUDGE_REQUEST_FAILED", "LLM judge gateway request failed", { status: response.status, @@ -1089,6 +1088,40 @@ async function requestJudgeOnce(request: JudgeGatewayRequest, input: FindingJudg } } +const MAX_JUDGE_RESPONSE_BYTES = 1024 * 1024; + +async function boundedJudgeResponseText(response: Response): Promise { + const tooLarge = (): EvalError => + new EvalError( + "EVAL_LLM_JUDGE_RESPONSE_TOO_LARGE", + `LLM judge response exceeded ${String(MAX_JUDGE_RESPONSE_BYTES)} bytes`, + { label: "LLM judge", maxBytes: MAX_JUDGE_RESPONSE_BYTES } + ); + const declaredBytes = Number(response.headers.get("content-length") ?? 0); + if (Number.isFinite(declaredBytes) && declaredBytes > MAX_JUDGE_RESPONSE_BYTES) { + await response.body?.cancel(); + throw tooLarge(); + } + if (response.body === null) return await response.text(); + + const chunks: Uint8Array[] = []; + let totalBytes = 0; + const reader = (response.body as ReadableStream).getReader(); + try { + for (let read = await reader.read(); !read.done; read = await reader.read()) { + totalBytes += read.value.byteLength; + if (totalBytes > MAX_JUDGE_RESPONSE_BYTES) { + await reader.cancel(); + throw tooLarge(); + } + chunks.push(read.value); + } + } finally { + reader.releaseLock(); + } + return Buffer.concat(chunks, totalBytes).toString("utf8"); +} + function parseJudgeCompletion(bodyText: string, input: FindingJudgeInput): FindingJudgeResult { let content = ""; try { diff --git a/packages/evals/src/types.ts b/packages/evals/src/types.ts index 2b0016662..fcf8d09fb 100644 --- a/packages/evals/src/types.ts +++ b/packages/evals/src/types.ts @@ -7,7 +7,6 @@ export const EVAL_RUN_SUMMARY_SCHEMA_VERSION = "ultrafuzz.eval.run-summary.v2" a export const EVAL_FINDING_SCORE_SCHEMA_VERSION = "ultrafuzz.eval.finding-score.v2" as const; export const EVAL_SCORE_SUMMARY_SCHEMA_VERSION = "ultrafuzz.eval.score-summary.v2" as const; export const EVAL_REVIEW_QUEUE_ITEM_SCHEMA_VERSION = "ultrafuzz.eval.review-queue-item.v2" as const; -export const EVAL_PUBLICATION_STATE_SCHEMA_VERSION = "ultrafuzz.eval.publication.v1" as const; export type EvalClassification = "true-positive" | "false-positive" | "needs-human-review" | "missed"; export type EvalClassificationReasonCode = @@ -95,7 +94,10 @@ export interface EvalTarget { signal_profile?: string; /** Relative ground-truth file resolved strictly under the operator-supplied `[eval].ground_truth_root`. */ ground_truth: string; - /** `private` forces manifest-only artifact reporting unless the suite explicitly opts into `upload`. */ + /** + * `private` requires the ground truth to be bound to this repo and ref, and + * `ULTRAFUZZ_EVAL_JUDGE_ALLOW_PRIVATE_DATA=true` before an LLM judge receives it. + */ sensitivity?: "public" | "private"; /** * Benchmark paths the run must never read, such as a reference solution the @@ -166,17 +168,15 @@ export interface EvalRecoveryEquivalence { export type EvalArtifactMode = "manifest-only" | "upload"; +/** + * The suite's `reporting.artifacts` block: validated and recorded, but no eval + * behaviour depends on it. `mode` defaults to `manifest-only`. + */ export interface EvalArtifactPolicy { - /** - * `manifest-only` publishes file names/sizes/hashes only; `upload` also streams payloads. - * `manifest-only` is the default and is always forced for `sensitivity: private` - * targets unless the suite explicitly sets `upload` (the opt-in). - */ mode: EvalArtifactMode; - /** Allowlist of artifact file names eligible for streaming to the provider. */ include: string[]; max_file_bytes: number; - /** True when the suite YAML explicitly set `mode` (the privacy opt-in signal). */ + /** True when the suite YAML explicitly set `mode`. */ mode_explicit: boolean; } @@ -689,20 +689,6 @@ export interface HumanReviewQueueItem { reviewer_status: ReviewerStatus; } -export interface EvalPublicationDiagnostic { - code: "TERMINAL_REPORT_NOT_PUBLISHABLE" | "RECOVERY_EQUIVALENCE_NOT_PUBLISHABLE"; - row_id: string; - contract: "ultrafuzz/report@3"; - reason: string; - report_path?: string; -} - -export interface EvalPublicationState { - schema_version: typeof EVAL_PUBLICATION_STATE_SCHEMA_VERSION; - status: "publishable" | "non-publishable"; - diagnostics: EvalPublicationDiagnostic[]; -} - export interface EvalRowScore { row_id: string; target_id: string; diff --git a/packages/evals/src/utils.ts b/packages/evals/src/utils.ts index a8f441b30..8765c2af4 100644 --- a/packages/evals/src/utils.ts +++ b/packages/evals/src/utils.ts @@ -107,10 +107,6 @@ function truncateUtf8(value: string, maxBytes: number, marker: string): string { return `${value.slice(0, end)}${marker}`; } -export function warningDiagnostic(code: string, message: string): RuntimeDiagnostic { - return { code, message, severity: "warning", source: "evals" }; -} - export function assertSafeEvalId(value: string, label: string): string { try { return validateSafeId(value, label); @@ -390,36 +386,6 @@ export function mean(values: number[]): number { return roundMetric(values.reduce((sum, value) => sum + value, 0) / values.length); } -/** Deterministic RFC-4122-shaped UUID derived from stable parts (for provider run/span ids). */ -export function deterministicUuid(parts: string[]): string { - const digest = crypto.createHash("sha256").update(parts.join("\u0000")).digest("hex"); - return [ - digest.slice(0, 8), - digest.slice(8, 12), - `4${digest.slice(13, 16)}`, - `8${digest.slice(17, 20)}`, - digest.slice(20, 32) - ].join("-"); -} - -export function contentTypeForArtifact(relativePath: string): string { - const extension = path.extname(relativePath).toLowerCase(); - switch (extension) { - case ".md": - return "text/markdown"; - case ".json": - return "application/json"; - case ".txt": - case ".log": - return "text/plain"; - case ".yml": - case ".yaml": - return "application/yaml"; - default: - return "application/octet-stream"; - } -} - function resolveGitRef(targetPath: string, ref: string): string | undefined { try { return git(targetPath, ["rev-parse", "--verify", `${ref}^{commit}`]); diff --git a/packages/evals/test/eval-durable.test.ts b/packages/evals/test/eval-durable.test.ts index e420f0350..912457a88 100644 --- a/packages/evals/test/eval-durable.test.ts +++ b/packages/evals/test/eval-durable.test.ts @@ -10,16 +10,13 @@ import { describe, expect, it, vi } from "vitest"; import { appendEvalRunRecord, parseEvalFindingScore, - parseEvalPublicationState, parseEvalReviewQueueItem, parseEvalRunManifest, parseEvalRunRecord, parseEvalRunSummary, parseEvalScoreSummary, - parseTelemetryCursor, readEvalFindingScores, readEvalMatrix, - readEvalPublicationState, readEvalReviewQueue, readEvalRunManifest, readEvalRunRecords, @@ -29,7 +26,6 @@ import { serializeEvalFindingScores, serializeEvalReviewQueue, writeEvalMatrix, - writeEvalPublicationState, writeEvalRunManifest, writeEvalRunSummary, writeEvalScoreSummary @@ -237,24 +233,6 @@ describe("eval durable schema registry", () => { for (const [schemaId, value] of fixtures) { expect(validateEvalJsonSchema(schemaId, value), schemaId).toMatchObject({ ok: true, issues: [] }); } - expect( - parseEvalPublicationState({ - schema_version: "ultrafuzz.eval.publication.v1", - status: "publishable", - diagnostics: [] - }) - ).toBeDefined(); - expect( - parseTelemetryCursor({ - schemaVersion: "ultrafuzz.eval.telemetry-cursor.v1", - byteOffset: 0, - deliveredEventIds: [], - uploadedArtifacts: {}, - lastHeartbeatAt: {}, - providerIds: {}, - findingsCountByNode: {} - }) - ).toBeDefined(); }); it("keeps durable run-manifest concurrency limits aligned with suite planning", () => { @@ -294,24 +272,17 @@ describe("eval durable readers and writers", () => { const recordsPath = path.join(root, "runs.jsonl"); const runSummaryPath = path.join(root, "run-summary.json"); const scoreSummaryPath = path.join(root, "summary.json"); - const publicationPath = path.join(root, "publication-state.json"); writeEvalRunManifest(manifestPath, manifest); writeEvalMatrix(matrixPath, [row]); fs.writeFileSync(recordsPath, ""); appendEvalRunRecord(recordsPath, record); writeEvalRunSummary(runSummaryPath, runSummary); writeEvalScoreSummary(scoreSummaryPath, scoreSummary); - writeEvalPublicationState(publicationPath, { - schema_version: "ultrafuzz.eval.publication.v1", - status: "publishable", - diagnostics: [] - }); expect(readEvalRunManifest(manifestPath)).toEqual(manifest); expect(readEvalMatrix(matrixPath)).toEqual([row]); expect(readEvalRunRecords(recordsPath)).toEqual([record]); expect(readEvalRunSummary(runSummaryPath)).toEqual(runSummary); expect(readEvalScoreSummary(scoreSummaryPath)).toEqual(scoreSummary); - expect(readEvalPublicationState(publicationPath).status).toBe("publishable"); }); it("rejects duplicate keys, invalid UTF-8, and wrong versions", () => { diff --git a/packages/evals/test/helpers.ts b/packages/evals/test/helpers.ts index b03cabe53..6a638e90a 100644 --- a/packages/evals/test/helpers.ts +++ b/packages/evals/test/helpers.ts @@ -30,15 +30,6 @@ import { } from "@ultrafuzz/artifacts"; import { WORKFLOW_CONTROL_INTEGRITY_SCHEMA_VERSION, projectCanonicalFinalReport } from "@ultrafuzz/runtime"; -import type { - EvalArtifactUpload, - EvalNodeEventEnvelope, - EvalPlan, - EvalReporter, - EvalRowGraph, - EvalRowResult, - EvalSummary -} from "../src/reporter.js"; import type { EvalMatrixRow, EvalRunProvenance, @@ -186,7 +177,7 @@ export function recoveryEquivalenceSummary( }; } -export function testReportingPolicy(overrides: Partial = {}): EvalReportingPolicy { +function testReportingPolicy(overrides: Partial = {}): EvalReportingPolicy { return { node_telemetry: true, heartbeat_interval_seconds: 60, @@ -1141,69 +1132,6 @@ function verifiedFinalReportOutputs(reportJsonRelativePath = "report.json"): Art ]; } -export interface RecordedCall { - method: string; - args: unknown[]; -} - -/** In-memory reporter that records every callback for exact-sequence assertions. */ -export class RecordingReporter implements EvalReporter { - readonly name: string; - readonly calls: RecordedCall[] = []; - failOn: Set = new Set(); - - constructor(name = "recording") { - this.name = name; - } - - envelopes(): EvalNodeEventEnvelope[] { - return this.calls - .filter((call) => call.method === "onNodeEvent") - .map((call) => call.args[0] as EvalNodeEventEnvelope); - } - - artifacts(): EvalArtifactUpload[] { - return this.calls.filter((call) => call.method === "onArtifact").map((call) => call.args[0] as EvalArtifactUpload); - } - - private record(method: string, args: unknown[]): Promise { - if (this.failOn.has(method)) { - return Promise.reject(new Error(`${method} forced failure`)); - } - this.calls.push({ method, args }); - return Promise.resolve(); - } - - onPlan(plan: EvalPlan): Promise { - return this.record("onPlan", [plan]); - } - - onRowStart(row: EvalMatrixRow, graph: EvalRowGraph): Promise { - return this.record("onRowStart", [row, graph]); - } - - onNodeEvent(envelope: EvalNodeEventEnvelope): Promise { - return this.record("onNodeEvent", [envelope]); - } - - onArtifact(artifact: EvalArtifactUpload): Promise { - return this.record("onArtifact", [artifact]); - } - - onRowFinish(row: EvalMatrixRow, result: EvalRowResult): Promise { - return this.record("onRowFinish", [row, result]); - } - - onScores(scores: EvalRowScore[], summary: EvalSummary): Promise { - return this.record("onScores", [scores, summary]); - } - - async finalize(summary: EvalSummary): Promise<{ url?: string }> { - await this.record("finalize", [summary]); - return { url: "https://example.com/experiment" }; - } -} - type JournalEventInputFor = RecordType extends EventRecord ? Omit : never; diff --git a/packages/evals/test/history-publication.test.ts b/packages/evals/test/history-publication.test.ts deleted file mode 100644 index 9b14be888..000000000 --- a/packages/evals/test/history-publication.test.ts +++ /dev/null @@ -1,197 +0,0 @@ -import fs from "node:fs"; -import os from "node:os"; -import path from "node:path"; - -import { describe, expect, it } from "vitest"; - -import { - EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_VERSION, - EVAL_HISTORY_PUBLICATION_HANDOFF_GATE, - EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_VERSION, - assertEvalHistoryPublicationHandoff, - evalHistoryPublicationHandoffIssues, - parseEvalHistoryAutomaticPublicationPlan, - parseEvalHistoryPublicationGeneration, - readEvalHistoryAutomaticPublicationPlan, - readEvalHistoryPublicationGeneration, - type EvalHistoryAutomaticPublicationPlan, - type EvalHistoryPublicationGeneration -} from "../src/history-publication.js"; -import { - EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_ID, - EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_ID, - validateEvalJsonSchema -} from "../src/eval-schema-registry.js"; -import { executeEvalSchemaSemanticGates } from "../src/eval-semantic-gates.js"; - -const REPOSITORY = "https://github.com/monad-developers/ultrafuzz"; -const SOURCE_ARTIFACT = `${REPOSITORY}/actions/runs/12345`; -const PUBLICATION_URL = `${SOURCE_ARTIFACT}/artifacts`; -const MODEL_SLUG = "benchmark-smoke-gpt-5-6-luna-high"; -const PAIR = `ultrafuzz-bench-${MODEL_SLUG}`; -const TARGET_IDS = ["target-a", "target-b", "target-c"]; - -function publicationPlan(): EvalHistoryAutomaticPublicationPlan { - return { - schema_version: EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_VERSION, - candidate_commit: "a".repeat(40), - candidate_repository_url: REPOSITORY, - source_artifact: SOURCE_ARTIFACT, - producer_run_id: "12345", - producer_run_attempt: "2", - mode: "smoke", - benchmark: "ultrafuzz-bench", - pairs: [ - { - pair: PAIR, - provider: "openai", - model_slug: MODEL_SLUG, - bundle_path: `${PAIR}/${MODEL_SLUG}/public-results.json`, - unpack_path: PAIR, - eval_run_id: "eval-run-1", - benchmark: "ultrafuzz-bench", - lane: "smoke", - status: "succeeded", - target_ids: TARGET_IDS, - executed_case_count: 3, - graded_case_count: 3, - publication_url: PUBLICATION_URL - } - ] - }; -} - -function publicationGeneration(): EvalHistoryPublicationGeneration { - return { - schema_version: EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_VERSION, - candidate_commit: "a".repeat(40), - candidate_repository_url: REPOSITORY, - source_artifact: SOURCE_ARTIFACT, - runs: [ - { - eval_run_id: "eval-run-1", - benchmark: "ultrafuzz-bench", - lane: "smoke", - status: "succeeded", - input_path: `${PAIR}/eval`, - target_ids: TARGET_IDS, - executed_case_count: 3, - graded_case_count: 3, - publication_url: PUBLICATION_URL - } - ] - }; -} - -describe("eval-history publication documents", () => { - it("registers exact current-only plan and generation shapes with executable integrity gates", () => { - const plan = publicationPlan(); - const generation = publicationGeneration(); - expect(validateEvalJsonSchema(EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_ID, plan).ok).toBe(true); - expect(validateEvalJsonSchema(EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_ID, generation).ok).toBe(true); - expect(executeEvalSchemaSemanticGates(EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_ID, plan)).toEqual([]); - expect(executeEvalSchemaSemanticGates(EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_ID, generation)).toEqual([]); - expect(parseEvalHistoryAutomaticPublicationPlan(plan)).toBe(plan); - expect(parseEvalHistoryPublicationGeneration(generation)).toBe(generation); - }); - - it("requires enriched generation rows and rejects aliases or synthesized publication URLs", () => { - const missingStatus = structuredClone(publicationGeneration()) as unknown as Record & { - runs: Array>; - }; - delete missingStatus.runs[0]!.status; - expect(() => parseEvalHistoryPublicationGeneration(missingStatus)).toThrowError( - expect.objectContaining({ code: "EVAL_HISTORY_PUBLICATION_GENERATION_INVALID" }) - ); - - const missingPublicationUrl = structuredClone(publicationGeneration()) as unknown as Record & { - runs: Array>; - }; - delete missingPublicationUrl.runs[0]!.publication_url; - expect(() => parseEvalHistoryPublicationGeneration(missingPublicationUrl)).toThrow(); - - expect(() => - parseEvalHistoryPublicationGeneration({ - ...publicationGeneration(), - candidate_repository_url: `${REPOSITORY}/` - }) - ).toThrow(); - }); - - it("rejects cross-field identity drift through the named semantic gates", () => { - const plan = { ...publicationPlan(), source_artifact: `${REPOSITORY}/actions/runs/99999` }; - expect(validateEvalJsonSchema(EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_ID, plan).ok).toBe(true); - expect(executeEvalSchemaSemanticGates(EVAL_HISTORY_AUTOMATIC_PUBLICATION_PLAN_SCHEMA_ID, plan)).toEqual( - expect.arrayContaining([ - expect.objectContaining({ - gate: "eval-history-automatic-publication-plan-integrity", - path: "$.source_artifact" - }) - ]) - ); - expect(() => parseEvalHistoryAutomaticPublicationPlan(plan)).toThrow(); - - const generation = structuredClone(publicationGeneration()); - generation.runs[0]!.publication_url = `${REPOSITORY}/actions/runs/99999/artifacts`; - expect(validateEvalJsonSchema(EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_ID, generation).ok).toBe(true); - expect(executeEvalSchemaSemanticGates(EVAL_HISTORY_PUBLICATION_GENERATION_SCHEMA_ID, generation)).toEqual([ - expect.objectContaining({ - gate: "eval-history-publication-generation-integrity", - path: "$.runs[0].publication_url" - }) - ]); - expect(() => parseEvalHistoryPublicationGeneration(generation)).toThrow(); - }); - - it("executes a named exact join between each plan pair and generation run", () => { - const plan = publicationPlan(); - const generation = publicationGeneration(); - expect(evalHistoryPublicationHandoffIssues(plan, generation)).toEqual([]); - expect(() => assertEvalHistoryPublicationHandoff(plan, generation)).not.toThrow(); - - const drifted = structuredClone(generation); - drifted.runs[0]!.input_path = "other/eval"; - expect(evalHistoryPublicationHandoffIssues(plan, drifted)).toEqual([ - { - gate: EVAL_HISTORY_PUBLICATION_HANDOFF_GATE, - path: "$.generation.runs[0]", - message: "must equal the canonical projection of plan.pairs[0]" - } - ]); - expect(() => assertEvalHistoryPublicationHandoff(plan, drifted)).toThrowError( - expect.objectContaining({ code: "EVAL_HISTORY_PUBLICATION_HANDOFF_INVALID" }) - ); - }); - - it.runIf(process.platform !== "win32")( - "strict-reads immutable regular files and rejects duplicate keys and symlinks", - () => { - const root = fs.mkdtempSync(path.join(fs.realpathSync(os.tmpdir()), "ultrafuzz-history-publication-")); - try { - const planPath = path.join(root, "plan.json"); - const generationPath = path.join(root, "generation.json"); - fs.writeFileSync(planPath, `${JSON.stringify(publicationPlan())}\n`, "utf8"); - fs.writeFileSync(generationPath, `${JSON.stringify(publicationGeneration())}\n`, "utf8"); - expect(readEvalHistoryAutomaticPublicationPlan(planPath)).toEqual(publicationPlan()); - expect(readEvalHistoryPublicationGeneration(generationPath)).toEqual(publicationGeneration()); - - const duplicatePath = path.join(root, "duplicate.json"); - fs.writeFileSync( - duplicatePath, - `${JSON.stringify(publicationGeneration()).replace( - '"schema_version":"ultrafuzz.eval-history-publication-generation.v1"', - '"schema_version":"ultrafuzz.eval-history-publication-generation.v1","schema_version":"shadow"' - )}\n`, - "utf8" - ); - expect(() => readEvalHistoryPublicationGeneration(duplicatePath)).toThrow(/duplicate property/u); - - const linkPath = path.join(root, "generation-link.json"); - fs.symlinkSync(generationPath, linkPath); - expect(() => readEvalHistoryPublicationGeneration(linkPath)).toThrow(); - } finally { - fs.rmSync(root, { recursive: true, force: true }); - } - } - ); -}); diff --git a/packages/evals/test/history-recency.test.ts b/packages/evals/test/history-recency.test.ts deleted file mode 100644 index 088b1e483..000000000 --- a/packages/evals/test/history-recency.test.ts +++ /dev/null @@ -1,185 +0,0 @@ -import path from "node:path"; -import { fileURLToPath } from "node:url"; - -import { describe, expect, it } from "vitest"; - -import { - EVAL_HISTORY_OBSERVATION_SCHEMA_VERSION, - assertEvalHistoryRecency, - emptyEvalHistory, - mergeEvalHistory, - readEvalHistory, - type EvalHistoryObservation -} from "../src/history.js"; - -const REPOSITORY_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../../.."); -const CANDIDATE = "1111111111111111111111111111111111111111"; -const TARGET_REVISION = "2222222222222222222222222222222222222222"; -const FINGERPRINT = `sha256:${"a".repeat(64)}`; -const EXECUTION_POLICY_FINGERPRINT = `sha256:${"d".repeat(64)}`; -const SCORING_FINGERPRINT = `sha256:${"b".repeat(64)}`; -const PUBLICATION_URL = "https://github.com/monad-developers/ultrafuzz/actions/runs/123/artifacts"; -const MILLISECONDS_PER_DAY = 86_400_000; - -function observation(overrides: Partial = {}): EvalHistoryObservation { - return { - schema_version: EVAL_HISTORY_OBSERVATION_SCHEMA_VERSION, - id: "run-1:target-a:baseline:benchmark-smoke", - benchmark: "evmbench", - lane: "smoke", - status: "succeeded", - target: "target-a", - variant: "baseline", - trial_count: 1, - run_timestamp: "2026-07-19T00:00:00.000Z", - candidate_commit: CANDIDATE, - candidate_repository_url: "https://github.com/monad-developers/ultrafuzz", - cohort_fingerprint: FINGERPRINT, - target_revisions: [{ target: "target-a", revision: TARGET_REVISION }], - model_profile: "benchmark-smoke", - model: "gpt-5.6-luna", - reasoning_effort: "high", - execution_policy_fingerprint: EXECUTION_POLICY_FINGERPRINT, - scoring_fingerprint: SCORING_FINGERPRINT, - precision: 0.75, - recall: 0.5, - f1: 0.6, - cumulative_unique_true_positives: 2, - ground_truth_bug_count: 2, - wall_clock_seconds: 120, - wall_clock_completeness: { status: "complete", reasons: [] }, - cost_usd: 1.25, - cost_completeness: { status: "complete", reasons: [] }, - executed_case_count: 1, - graded_case_count: 1, - publication_url: PUBLICATION_URL, - target_publication: { - target: "target-a", - repository: "https://example.com/target-a", - revision: TARGET_REVISION, - status: "succeeded", - executed_case_count: 1, - graded_case_count: 1, - publication_location: { - bundle_path: "public-results.json", - report_paths: ["reports/target-a-baseline-trial-1/report.md", "reports/target-a-baseline-trial-1/report.json"] - } - }, - source_eval_run_id: "run-1", - source_artifact: "https://github.com/monad-developers/ultrafuzz/actions/runs/1", - ...overrides - }; -} - -describe("eval history recency", () => { - // `now` is derived from the newest observation, so this pins the comparison and the - // rendered message against a synthetic clock; it says nothing about how stale the - // checked-in history actually is. Deriving the clock keeps the expected message from - // rotting as wall-clock time advances, and detecting a real outage belongs to the - // caller that chooses the maximum. - it("renders the staleness verdict and message for the checked-in history against a clock 19.5 days past its newest observation", () => { - const history = readEvalHistory(path.join(REPOSITORY_ROOT, "benchmarks", "ultrafuzzbench", "history.json")); - const [first, ...rest] = history.observations; - if (first === undefined) throw new Error("checked-in eval history has no observation to age"); - const newest = rest.reduce( - (latest, candidate) => (candidate.run_timestamp > latest ? candidate.run_timestamp : latest), - first.run_timestamp - ); - const now = new Date(Date.parse(newest) + 19.5 * MILLISECONDS_PER_DAY); - - expect(() => assertEvalHistoryRecency({ history, maxAgeDays: 30, now })).not.toThrow(); - expect(() => assertEvalHistoryRecency({ history, maxAgeDays: 7, now })).toThrowError( - expect.objectContaining({ - code: "EVAL_HISTORY_STALE", - message: `newest eval history observation ran at ${newest}, 19.5 days ago, exceeding the requested 7 day maximum`, - details: { newest_run_timestamp: newest, age_days: 19.5, max_age_days: 7 } - }) - ); - }); - - it("ages an unordered history from its newest observation and refuses to age an empty one", () => { - const newest = "2026-07-25T00:00:00.000Z"; - const unordered = mergeEvalHistory(emptyEvalHistory(), [ - observation({ id: "run-1:target-a:baseline:benchmark-smoke", run_timestamp: "2026-07-19T00:00:00.000Z" }), - observation({ - id: "run-2:target-a:baseline:benchmark-smoke", - run_timestamp: newest, - source_eval_run_id: "run-2" - }), - observation({ id: "run-3:target-a:baseline:benchmark-smoke", run_timestamp: "2026-07-21T00:00:00.000Z" }) - ]); - const now = new Date("2026-07-28T00:00:00.000Z"); - - expect(() => assertEvalHistoryRecency({ history: unordered, maxAgeDays: 3, now })).not.toThrow(); - expect(() => assertEvalHistoryRecency({ history: unordered, maxAgeDays: 2, now })).toThrowError( - expect.objectContaining({ - code: "EVAL_HISTORY_STALE", - message: `newest eval history observation ran at ${newest}, 3 days ago, exceeding the requested 2 day maximum` - }) - ); - expect(() => assertEvalHistoryRecency({ history: emptyEvalHistory(), maxAgeDays: 7, now })).toThrowError( - expect.objectContaining({ - code: "EVAL_HISTORY_EMPTY", - message: "eval history has no observation to age against the requested 7 day maximum" - }) - ); - }); - - it("refuses to age a history whose newest observation is dated in the future", () => { - const future = "2027-08-31T00:00:00.000Z"; - const history = mergeEvalHistory(emptyEvalHistory(), [ - observation({ id: "run-1:target-a:baseline:benchmark-smoke", run_timestamp: "2026-08-30T00:00:00.000Z" }), - observation({ - id: "run-2:target-a:baseline:benchmark-smoke", - run_timestamp: future, - source_eval_run_id: "run-2" - }) - ]); - const now = new Date("2026-08-31T00:00:00.000Z"); - - expect(() => assertEvalHistoryRecency({ history, maxAgeDays: 2, now })).toThrowError( - expect.objectContaining({ - code: "EVAL_HISTORY_FUTURE_DATED", - message: `newest eval history observation ran at ${future}, which is in the future at 2026-08-31T00:00:00.000Z, so its age cannot be measured against the requested 2 day maximum`, - details: { newest_run_timestamp: future, now: "2026-08-31T00:00:00.000Z", max_age_days: 2 } - }) - ); - expect(() => assertEvalHistoryRecency({ history, maxAgeDays: 10_000, now })).toThrowError( - expect.objectContaining({ code: "EVAL_HISTORY_FUTURE_DATED" }) - ); - }); - - it("reports an age just past the maximum with more precision than the maximum", () => { - const newest = "2026-08-28T23:55:00.000Z"; - const history = mergeEvalHistory(emptyEvalHistory(), [observation({ run_timestamp: newest })]); - const now = new Date("2026-08-31T00:00:00.000Z"); - - expect(() => assertEvalHistoryRecency({ history, maxAgeDays: 2, now })).toThrowError( - expect.objectContaining({ - code: "EVAL_HISTORY_STALE", - message: `newest eval history observation ran at ${newest}, 2.0035 days ago, exceeding the requested 2 day maximum` - }) - ); - }); - - it("refuses to age a history against an unusable maximum or clock", () => { - const only = observation(); - const history = mergeEvalHistory(emptyEvalHistory(), [only]); - const now = new Date("2026-07-19T00:00:00.000Z"); - for (const maxAgeDays of [0, -1, Number.NaN, Number.POSITIVE_INFINITY]) { - expect(() => assertEvalHistoryRecency({ history, maxAgeDays, now })).toThrowError( - expect.objectContaining({ code: "EVAL_HISTORY_RECENCY_INVALID" }) - ); - } - expect(() => assertEvalHistoryRecency({ history, maxAgeDays: 7, now: new Date(Number.NaN) })).toThrowError( - expect.objectContaining({ code: "EVAL_HISTORY_RECENCY_INVALID" }) - ); - expect(() => - assertEvalHistoryRecency({ - history: { ...history, observations: [{ ...only, run_timestamp: "2026-02-30T00:00:00.000Z" }] }, - maxAgeDays: 7, - now - }) - ).toThrowError(expect.objectContaining({ code: "EVAL_HISTORY_INVALID" })); - }); -}); diff --git a/packages/evals/test/history.test.ts b/packages/evals/test/history.test.ts index f62792267..0354949f0 100644 --- a/packages/evals/test/history.test.ts +++ b/packages/evals/test/history.test.ts @@ -500,9 +500,6 @@ describe("longitudinal eval history", () => { expect(pendingCharts.get("performance-cost.svg")).toContain( 'data-model="deepseek-v4-flash" data-status="partial" data-run-count="1" data-expected-run-count="1" data-available-target-count="3" data-expected-target-count="3"' ); - expect(pendingCharts.get("cost.svg")).toContain( - 'data-status="partial" data-available-count="1" data-expected-count="1"' - ); const replacement = supersessionRun({ sourceRunId: replacementSourceRun, @@ -515,9 +512,6 @@ describe("longitudinal eval history", () => { expect(renderEvalHistoryCharts(partiallyMerged).get("performance-cost.svg")).toContain( 'data-model="deepseek-v4-flash" data-status="partial" data-run-count="1" data-expected-run-count="1" data-available-target-count="3" data-expected-target-count="3"' ); - expect(renderEvalHistoryCharts(partiallyMerged).get("cost.svg")).toContain( - 'data-status="partial" data-available-count="1" data-expected-count="1"' - ); const merged = mergeEvalHistory(partiallyMerged, replacement); expect(merged.supersessions).toEqual([supersession]); @@ -529,10 +523,8 @@ describe("longitudinal eval history", () => { ); expect(performance).not.toContain('data-available-target-count="5" data-expected-target-count="6"'); expect(performance).not.toContain("targets 5/6"); - expect(replacedCharts.get("cost.svg")).not.toContain("7777777"); - expect(replacedCharts.get("cost.svg")).toContain("8888888"); - expect(replacedCharts.get("precision.svg")).not.toContain(`/commit/${"7".repeat(40)}`); - expect(replacedCharts.get("precision.svg")).toContain(`/commit/${"8".repeat(40)}`); + expect(replacedCharts.get("quality.svg")).not.toContain(`/commit/${"7".repeat(40)}`); + expect(replacedCharts.get("quality.svg")).toContain(`/commit/${"8".repeat(40)}`); }); it("activates an explicit cohort transition only when both immutable fingerprints match", () => { @@ -567,15 +559,13 @@ describe("longitudinal eval history", () => { }); const partiallyMerged = mergeEvalHistory(pending, replacement.slice(0, 2)); - const partialCost = renderEvalHistoryCharts(partiallyMerged).get("cost.svg")!; - expect(partialCost).toContain('data-completeness-marker="partial"'); - expect(partialCost).toContain("0.2 partial (pricing-incomplete)"); - expect(partialCost).not.toContain("partial n/a 7777777"); + expect(renderEvalHistoryCharts(partiallyMerged).get("quality.svg")).toContain(`/commit/${"7".repeat(40)}`); const merged = mergeEvalHistory(partiallyMerged, replacement); expect(merged.supersessions).toEqual([supersession]); - expect(renderEvalHistoryCharts(merged).get("cost.svg")).not.toContain("7777777"); - expect(renderEvalHistoryCharts(merged).get("precision.svg")).toContain(`/commit/${"8".repeat(40)}`); + const mergedCharts = renderEvalHistoryCharts(merged); + expect(mergedCharts.get("quality.svg")).not.toContain(`/commit/${"7".repeat(40)}`); + expect(mergedCharts.get("quality.svg")).toContain(`/commit/${"8".repeat(40)}`); }); it("rejects invalid source-run supersession ledgers", () => { @@ -959,11 +949,6 @@ describe("longitudinal eval history", () => { expect(charts.get("performance-cost.svg")).toContain( 'deepseek-v4-flash · median 23.7% · $0.41 · n=1 · targets 2/3 · partial' ); - expect(charts.get("cost.svg")).toContain( - 'data-status="unavailable" data-available-count="0" data-expected-count="1"' - ); - expect(charts.get("cost.svg")).not.toContain('data-completeness-marker="partial-null"'); - expect(charts.get("cost.svg")).toContain("n/a 70646d2"); expect(charts.get("latest-summary.svg")).toContain( 'data-metric="cost_usd" data-status="partial" data-available-target-count="2" data-expected-target-count="3"' ); @@ -1586,14 +1571,12 @@ describe("longitudinal eval history", () => { const first = renderEvalHistoryCharts(history); const second = renderEvalHistoryCharts(history); expect(first).toEqual(second); - expect(first.get("precision.svg")).toContain(`https://github.com/monad-developers/ultrafuzz/commit/${CANDIDATE}`); - expect(first.get("precision.svg")).toContain(CANDIDATE.slice(0, 7)); - expect(first.get("precision.svg")).toContain("gpt-5.6-luna · high"); - expect(first.get("precision.svg")).toContain("cohort-aaaaaaaa"); - expect(first.get("precision.svg")).toContain("policy-dddddddd"); - expect(first.get("precision.svg")).toContain(">target-a"); - expect(first.get("wall-clock-time.svg")).toContain('data-status="unavailable"'); - expect(first.get("wall-clock-time.svg")).toContain(`>n/a ${CANDIDATE.slice(0, 7)}<`); + expect([...first.keys()]).toEqual(["latest-summary.svg", "quality.svg", "performance-cost.svg"]); + expect(first.get("latest-summary.svg")).toContain( + `https://github.com/monad-developers/ultrafuzz/commit/${CANDIDATE}` + ); + expect(first.get("latest-summary.svg")).toContain("gpt-5.6-luna · high"); + expect(first.get("quality.svg")).toContain("cohort-aaaaaaaa"); expect(first.get("latest-summary.svg")).toContain("Score (macro-F1)"); expect(first.get("latest-summary.svg")).toContain("n/a · unavailable 0/1"); expect(first.get("quality.svg")).toContain('data-metric="f1"'); @@ -1818,11 +1801,6 @@ describe("longitudinal eval history", () => { expect(svg).toContain( 'deepseek-v4-flash · median 25.0% · $0.40 · n=1 · targets 2/2 · partial' ); - const costSvg = renderEvalHistoryCharts( - parseEvalHistory({ schema_version: EVAL_HISTORY_SCHEMA_VERSION, supersessions: [], observations: partialFlash }) - ).get("cost.svg")!; - expect(costSvg).toContain('data-status="partial" data-available-count="1" data-expected-count="1"'); - expect(costSvg).toContain("partial (pricing-incomplete)"); const deepseek = //u.exec(svg)?.[0]; expect(deepseek).toBeDefined(); expect(deepseek).not.toContain('data-iqr="'); @@ -1995,79 +1973,6 @@ describe("longitudinal eval history", () => { expect(formatted).toContain('"reasons": ["accounting-unavailable"]'); expect(parseEvalHistory(JSON.parse(formatted))).toEqual(history); }); - - it("renders changed cohorts and execution policies as lineage markers instead of chart series", () => { - const svg = renderEvalHistoryCharts( - parseEvalHistory({ - schema_version: EVAL_HISTORY_SCHEMA_VERSION, - supersessions: [], - observations: [ - observation({ id: "first-series" }), - observation({ - id: "second-series", - candidate_commit: "3333333333333333333333333333333333333333", - cohort_fingerprint: `sha256:${"c".repeat(64)}`, - execution_policy_fingerprint: `sha256:${"e".repeat(64)}` - }) - ] - }) - ).get("precision.svg")!; - - expect(svg.match(/cohort-aaaaaaaa policy-dddddddd target-a"); - expect(svg).not.toContain(">cohort-cccccccc policy-eeeeeeee target-a"); - }); - - it("labels the globally earliest and latest dates across series", () => { - const charts = renderEvalHistoryCharts( - parseEvalHistory({ - schema_version: EVAL_HISTORY_SCHEMA_VERSION, - supersessions: [], - observations: [ - observation({ id: "later", benchmark: "evmbench", run_timestamp: "2026-07-19T12:00:00.000Z" }), - observation({ - id: "earlier", - benchmark: "ultrafuzz-bench", - run_timestamp: "2026-07-17T12:00:00.000Z", - candidate_commit: "3333333333333333333333333333333333333333" - }) - ] - }) - ); - const svg = charts.get("precision.svg")!; - expect(svg.indexOf(">2026-07-17")).toBeLessThan(svg.indexOf(">2026-07-19")); - }); - - it("uses evenly spaced run columns and rotates date labels below the x axis", () => { - const observations = [0, 1, 2].map((index) => - observation({ - id: `run-${index}`, - candidate_commit: `${index + 1}`.repeat(40), - run_timestamp: ["2026-07-17T00:00:00.000Z", "2026-07-29T00:00:00.000Z", "2026-07-30T00:00:00.000Z"][index]!, - precision: 0.25 + index * 0.1 - }) - ); - const svg = renderEvalHistoryCharts( - parseEvalHistory({ - schema_version: EVAL_HISTORY_SCHEMA_VERSION, - supersessions: [], - observations - }) - ).get("precision.svg")!; - - const match = /]+points="([^"]+)"/u.exec(svg); - expect(match).not.toBeNull(); - const polylinePoints = match?.[1]; - expect(polylinePoints).toBeDefined(); - if (polylinePoints === undefined) throw new Error("missing polyline points"); - const xCoordinates = polylinePoints.split(" ").map((point) => Number(point.split(",")[0]!)); - expect(xCoordinates).toHaveLength(3); - expect(xCoordinates[1]! - xCoordinates[0]!).toBeCloseTo(xCoordinates[2]! - xCoordinates[1]!, 5); - expect(svg).toContain("rotate(-90)"); - expect(svg).toContain(">2026-07-29"); - }); }); function expectHistoryStructuralParity(value: unknown, expected: boolean): void { diff --git a/packages/evals/test/node-telemetry.test.ts b/packages/evals/test/node-telemetry.test.ts deleted file mode 100644 index 71963f919..000000000 --- a/packages/evals/test/node-telemetry.test.ts +++ /dev/null @@ -1,651 +0,0 @@ -import crypto from "node:crypto"; -import fs from "node:fs"; -import { mkdtempSync } from "node:fs"; -import { tmpdir } from "node:os"; -import path from "node:path"; - -import { describe, expect, it } from "vitest"; - -import { EVENT_SCHEMA_VERSION } from "@ultrafuzz/artifacts"; - -import { NodeTelemetryPump, loadTelemetryCursor } from "../src/node-telemetry.js"; -import { guardReporter, type EvalNodeEventEnvelope } from "../src/reporter.js"; -import { - currentPlannedGraph, - currentRunState, - RecordingReporter, - testReportingPolicy, - testRow, - testSuite, - writeVerifiedFinalReport, - writeRunFixture, - type JournalEventInput -} from "./helpers.js"; - -function setup(overrides: { policy?: ReturnType } = {}) { - const base = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ufz-evals-pump-")); - const runRoot = path.join(base, "run-1"); - const cursorPath = path.join(base, "cursor.json"); - const suite = testSuite(path.join(base, "gt")); - const row = testRow(suite, { run_id: "run-1" }); - const reporter = new RecordingReporter(); - const policy = overrides.policy ?? testReportingPolicy(); - const pump = () => - new NodeTelemetryPump({ - runRoot, - row, - reporters: [reporter], - policy, - cursorPath, - retryDelayMs: 0, - now: () => new Date("2026-07-09T00:10:00.000Z") - }); - return { base, runRoot, cursorPath, suite, row, reporter, pump, policy }; -} - -const T0 = "2026-07-09T00:00:00.000Z"; -const T1 = "2026-07-09T00:01:00.000Z"; -const T2 = "2026-07-09T00:05:00.000Z"; - -function eventId(label: string): string { - return `evt-${crypto.createHash("sha256").update(label).digest("hex").slice(0, 24)}`; -} - -function nodeSyncedEvent( - label: string, - timestamp: string, - status: "pending" | "running" | "succeeded" | "failed" | "timed-out" | "skipped", - attempt = 1 -): JournalEventInput { - return { - event_id: eventId(label), - event_type: "node-synced", - timestamp, - node_id: "setup-1", - status, - payload: { - workflow_run_id: "workflow-1", - workflow_task_id: "node:setup-1", - attempt - } - }; -} - -function manifestWrittenEvent(label: string, timestamp: string, fileCount = 1): JournalEventInput { - return { - event_id: eventId(label), - event_type: "artifact-manifest-written", - timestamp, - node_id: "setup-1", - status: "succeeded", - payload: { file_count: fileCount, path: "artifacts/setup-1/artifact-manifest.json" } - }; -} - -function completeEventRecord(event: JournalEventInput, runId = "run-1") { - return { schema_version: EVENT_SCHEMA_VERSION, run_id: runId, ...event }; -} - -describe("NodeTelemetryPump", () => { - it("translates journal records into exact envelope sequences", async () => { - const { runRoot, reporter, pump } = setup(); - writeRunFixture({ - runRoot, - events: [ - nodeSyncedEvent("sequence-started", T0, "running"), - { - event_id: eventId("sequence-findings"), - event_type: "findings-validated", - timestamp: T1, - node_id: "setup-1", - status: "succeeded", - payload: { count: 3, path: "artifacts/setup-1/deduped-findings.json" } - }, - manifestWrittenEvent("sequence-manifest", T1, 2), - nodeSyncedEvent("sequence-finished", T2, "succeeded"), - { - event_id: eventId("sequence-workflow"), - event_type: "workflow-synced", - timestamp: T2, - status: "running", - payload: { - workflow_run_id: "workflow-1", - workflow_status: "running", - workflow_state: "running", - synced_nodes: 1, - accounting_available: true, - recovery_due: false, - deadline_exceeded: false - } - } - ], - state: currentRunState({ - runId: "run-1", - status: "running", - nodes: { "setup-1": { started_at: T0, finished_at: T2 } }, - overrides: { created_at: T0, started_at: T0, last_transition_at: T2 } - }), - graph: currentPlannedGraph(["setup-1"], undefined), - artifacts: { - "setup-1": { - "report.md": "# report", - "not-allowlisted.bin": "xxx" - } - } - }); - - const result = await pump().drain(); - expect(result.warnings).toEqual([]); - const envelopes = reporter.envelopes(); - expect(envelopes.map((envelope) => envelope.event.type)).toEqual([ - "node-started", - "node-artifacts", - "node-finished" - ]); - expect(envelopes[0]).toMatchObject({ - eventId: eventId("sequence-started"), - rowId: "target-a-baseline-trial-1", - nodeId: "setup-1", - event: { type: "node-started", at: T0, attempt: 1 } - }); - expect(envelopes[0]?.idempotencyKey).toMatch(/^ultrafuzz-event-[0-9a-f]{64}$/u); - // manifest event announces both files (cheap, always sent) … - const artifactsEnvelope = envelopes[1]; - expect(artifactsEnvelope?.event).toMatchObject({ type: "node-artifacts" }); - const manifest = - artifactsEnvelope !== undefined && artifactsEnvelope.event.type === "node-artifacts" - ? artifactsEnvelope.event.manifest - : []; - expect(manifest.map((entry) => entry.path).sort()).toEqual(["not-allowlisted.bin", "report.md"]); - // … while onArtifact only streams the allowlisted file, without payload (manifest-only mode). - const uploads = reporter.artifacts(); - expect(uploads).toHaveLength(1); - expect(uploads[0]).toMatchObject({ - nodeId: "setup-1", - relativePath: "report.md", - contentType: "text/markdown" - }); - expect(uploads[0]?.idempotencyKey).toMatch(/^ultrafuzz-artifact-[0-9a-f]{64}$/u); - expect(uploads[0]?.read).toBeUndefined(); - // node-finished folds findingsCount and backdates startedAt from state.json. - expect(envelopes[2]).toMatchObject({ - eventId: eventId("sequence-finished"), - event: { type: "node-finished", status: "succeeded", at: T2, startedAt: T0, attempt: 1, findingsCount: 3 } - }); - }); - - it("streams payloads for allowlisted files when the suite opts into upload mode", async () => { - const policy = testReportingPolicy({ - artifacts: { mode: "upload", include: ["report.md"], max_file_bytes: 5_000_000, mode_explicit: true } - }); - const { runRoot, reporter, pump } = setup({ policy }); - writeRunFixture({ - runRoot, - events: [manifestWrittenEvent("stream-upload", T1)], - artifacts: { "setup-1": { "report.md": "# hello" } } - }); - await pump().drain(); - const uploads = reporter.artifacts(); - expect(uploads).toHaveLength(1); - expect(uploads[0]?.read).toBeDefined(); - const payload = await uploads[0]!.read!(); - expect(payload.toString("utf8")).toBe("# hello"); - }); - - it("rejects unsafe artifact manifest paths before upload", async () => { - const policy = testReportingPolicy({ - artifacts: { mode: "upload", include: ["report.md"], max_file_bytes: 5_000_000, mode_explicit: true } - }); - const { runRoot, reporter, pump } = setup({ policy }); - writeRunFixture({ - runRoot, - events: [manifestWrittenEvent("unsafe-manifest", T1)], - artifacts: { "setup-1": { "report.md": "# hello" } } - }); - const manifestPath = path.join(runRoot, "artifacts", "setup-1", "artifact-manifest.json"); - const manifest = JSON.parse(fs.readFileSync(manifestPath, "utf8")) as { - files: Array<{ path: string }>; - }; - manifest.files[0]!.path = "../../report.md"; - fs.writeFileSync(manifestPath, JSON.stringify(manifest), "utf8"); - - const result = await pump().drain(); - expect(reporter.artifacts()).toHaveLength(0); - expect(result.warnings).toEqual( - expect.arrayContaining([expect.objectContaining({ code: "EVAL_TELEMETRY_MANIFEST_UNREADABLE" })]) - ); - }); - - it("rejects allowlisted artifacts whose bytes do not match the manifest", async () => { - const policy = testReportingPolicy({ - artifacts: { mode: "upload", include: ["report.md"], max_file_bytes: 5_000_000, mode_explicit: true } - }); - const { runRoot, reporter, pump } = setup({ policy }); - writeRunFixture({ - runRoot, - events: [manifestWrittenEvent("artifact-digest", T1)], - artifacts: { "setup-1": { "report.md": "# hello" } } - }); - fs.writeFileSync(path.join(runRoot, "artifacts", "setup-1", "report.md"), "# changed", "utf8"); - - const result = await pump().drain(); - expect(reporter.artifacts()).toHaveLength(0); - expect(result.warnings).toEqual( - expect.arrayContaining([expect.objectContaining({ code: "EVAL_TELEMETRY_ARTIFACT_UNSAFE" })]) - ); - }); - - it("publishes no final-report payload after post-verification mutation", async () => { - const policy = testReportingPolicy({ - artifacts: { - mode: "upload", - include: ["report.md", "report.json"], - max_file_bytes: 5_000_000, - mode_explicit: true - } - }); - const { runRoot, reporter, pump } = setup({ policy }); - const verified = writeVerifiedFinalReport({ runRoot, runId: "run-1" }); - writeRunFixture({ - runRoot, - events: [ - { - event_id: eventId("verified-final-report-mutated"), - event_type: "artifact-manifest-written", - timestamp: T1, - node_id: "final-report", - status: "succeeded", - payload: { file_count: 2, path: "artifacts/final-report/artifact-manifest.json" } - } - ] - }); - fs.appendFileSync(verified.reportPath, " \n", "utf8"); - - const result = await pump().drain(); - - expect(reporter.artifacts()).toHaveLength(0); - expect(result.warnings).toEqual( - expect.arrayContaining([expect.objectContaining({ code: "EVAL_TELEMETRY_ARTIFACT_UNVERIFIED" })]) - ); - }); - - it("matches the artifact upload allowlist by exact relative path", async () => { - const policy = testReportingPolicy({ - artifacts: { mode: "upload", include: ["report.md"], max_file_bytes: 5_000_000, mode_explicit: true } - }); - const { runRoot, reporter, pump } = setup({ policy }); - writeRunFixture({ - runRoot, - events: [manifestWrittenEvent("exact-allowlist", T1)], - artifacts: { "setup-1": { "nested/report.md": "# nested" } } - }); - - await pump().drain(); - expect(reporter.artifacts()).toHaveLength(0); - }); - - it("keeps private targets manifest-only when upload mode was not explicit", async () => { - const policy = testReportingPolicy({ - artifacts: { mode: "upload", include: ["report.md"], max_file_bytes: 5_000_000, mode_explicit: false } - }); - const { runRoot, reporter, pump, row } = setup({ policy }); - row.target.sensitivity = "private"; - writeRunFixture({ - runRoot, - events: [manifestWrittenEvent("private-manifest", T1)], - artifacts: { "setup-1": { "report.md": "# hello" } } - }); - await pump().drain(); - // Row target is sensitivity: private in the fixture suite. - expect(reporter.artifacts()[0]?.read).toBeUndefined(); - }); - - it("does not double-publish across a simulated crash/resume", async () => { - const { runRoot, cursorPath, reporter, pump } = setup(); - writeRunFixture({ - runRoot, - events: [nodeSyncedEvent("resume-started", T0, "running"), nodeSyncedEvent("resume-finished", T1, "succeeded")] - }); - await pump().drain(); - expect(reporter.envelopes()).toHaveLength(2); - - // Simulate a crash: build a fresh pump from the persisted cursor and re-drain. - const resumed = pump(); - // Constructors do no unlocked cursor I/O; drain reloads authoritatively under the lease. - expect(resumed.cursor.byteOffset).toBe(0); - await resumed.drain(); - expect(resumed.cursor.byteOffset).toBeGreaterThan(0); - expect(reporter.envelopes()).toHaveLength(2); - - // Even replaying from offset 0 (rewritten cursor byteOffset) dedups by event_id. - const cursor = loadTelemetryCursor(cursorPath); - cursor.byteOffset = 0; - fs.writeFileSync(cursorPath, JSON.stringify(cursor), "utf8"); - await pump().drain(); - expect(reporter.envelopes()).toHaveLength(2); - - // An explicit publish reset is performed inside the same lease and replays - // with identical provider idempotency keys. - const originalKeys = reporter.envelopes().map((envelope) => envelope.idempotencyKey); - await pump().drain({ resetCursor: true }); - expect(reporter.envelopes()).toHaveLength(4); - expect( - reporter - .envelopes() - .slice(2) - .map((envelope) => envelope.idempotencyKey) - ).toEqual(originalKeys); - }); - - it("serializes concurrent drains across reload, callbacks, and durable commit", async () => { - const { runRoot, cursorPath, row, policy } = setup(); - writeRunFixture({ - runRoot, - events: [nodeSyncedEvent("concurrent", T0, "running")] - }); - - let callbackEntered!: () => void; - const entered = new Promise((resolve) => { - callbackEntered = resolve; - }); - let releaseCallback!: () => void; - const callbackGate = new Promise((resolve) => { - releaseCallback = resolve; - }); - class BlockingReporter extends RecordingReporter { - override async onNodeEvent(envelope: EvalNodeEventEnvelope): Promise { - await super.onNodeEvent(envelope); - callbackEntered(); - await callbackGate; - } - } - const reporter = new BlockingReporter(); - const makePump = () => - new NodeTelemetryPump({ runRoot, row, reporters: [reporter], policy, cursorPath, retryDelayMs: 0 }); - - const firstDrain = makePump().drain(); - await entered; - const secondDrain = makePump().drain(); - await new Promise((resolve) => setImmediate(resolve)); - expect(reporter.envelopes()).toHaveLength(1); - releaseCallback(); - - const results = await Promise.all([firstDrain, secondDrain]); - expect(results.reduce((total, result) => total + result.deliveredEvents, 0)).toBe(1); - expect(reporter.envelopes()).toHaveLength(1); - expect(loadTelemetryCursor(cursorPath).deliveredEventIds).toEqual([eventId("concurrent")]); - }); - - it("throws on cursor persistence failure and restores the in-memory durable snapshot", async () => { - const { runRoot, cursorPath, row, policy } = setup(); - writeRunFixture({ - runRoot, - events: [nodeSyncedEvent("write-failure", T0, "running")] - }); - class CursorBlockingReporter extends RecordingReporter { - override async onNodeEvent(envelope: EvalNodeEventEnvelope): Promise { - await super.onNodeEvent(envelope); - fs.mkdirSync(cursorPath); - } - } - const reporter = new CursorBlockingReporter(); - const telemetry = new NodeTelemetryPump({ - runRoot, - row, - reporters: [reporter], - policy, - cursorPath, - retryDelayMs: 0 - }); - - await expect(telemetry.drain()).rejects.toMatchObject({ code: "EVAL_TELEMETRY_CURSOR_WRITE_FAILED" }); - expect(reporter.envelopes()).toHaveLength(1); - expect(telemetry.cursor).toMatchObject({ byteOffset: 0, deliveredEventIds: [], lastHeartbeatAt: {} }); - expect(fs.existsSync(`${cursorPath}.tmp`)).toBe(false); - }); - - it("atomically replaces a cursor symlink introduced during delivery without following it", async () => { - const { base, runRoot, cursorPath, row, policy } = setup(); - writeRunFixture({ - runRoot, - events: [nodeSyncedEvent("symlink-swap", T0, "running")] - }); - const victimPath = path.join(base, "victim.txt"); - fs.writeFileSync(victimPath, "unchanged\n", "utf8"); - class SymlinkSwapReporter extends RecordingReporter { - override async onNodeEvent(envelope: EvalNodeEventEnvelope): Promise { - await super.onNodeEvent(envelope); - fs.symlinkSync(victimPath, cursorPath); - } - } - const reporter = new SymlinkSwapReporter(); - const telemetry = new NodeTelemetryPump({ - runRoot, - row, - reporters: [reporter], - policy, - cursorPath, - retryDelayMs: 0 - }); - - await telemetry.drain(); - expect(fs.lstatSync(cursorPath).isFile()).toBe(true); - expect(fs.lstatSync(cursorPath).isSymbolicLink()).toBe(false); - expect(fs.readFileSync(victimPath, "utf8")).toBe("unchanged\n"); - expect(loadTelemetryCursor(cursorPath).deliveredEventIds).toEqual([eventId("symlink-swap")]); - }); - - it("rejects malformed and dangling-symlink cursors instead of treating them as absent", () => { - const { base, cursorPath } = setup(); - fs.writeFileSync(cursorPath, '{"schemaVersion":', "utf8"); - expect(() => loadTelemetryCursor(cursorPath)).toThrow(); - - fs.unlinkSync(cursorPath); - fs.symlinkSync(path.join(base, "missing-cursor.json"), cursorPath); - expect(() => loadTelemetryCursor(cursorPath)).toThrow(/cannot be a symbolic link/u); - - const physicalDirectory = path.join(base, "physical-cursors"); - const linkedDirectory = path.join(base, "linked-cursors"); - fs.mkdirSync(physicalDirectory); - fs.symlinkSync(physicalDirectory, linkedDirectory); - expect(() => loadTelemetryCursor(path.join(linkedDirectory, "cursor.json"))).toThrow(/crosses symlink/u); - }); - - it("retries guarded reporter failures and leaves undelivered events for resume", async () => { - class FlakyReporter extends RecordingReporter { - attempts = 0; - remainingFailures = 2; - - override onNodeEvent(envelope: EvalNodeEventEnvelope): Promise { - this.attempts += 1; - if (this.remainingFailures > 0) { - this.remainingFailures -= 1; - return Promise.reject(new Error("temporary provider failure")); - } - return super.onNodeEvent(envelope); - } - } - - const { runRoot, cursorPath, row, policy } = setup(); - writeRunFixture({ - runRoot, - events: [nodeSyncedEvent("retry", T0, "running")] - }); - const reporter = new FlakyReporter(); - const guardWarnings: Array<{ code: string }> = []; - const guarded = guardReporter(reporter, (warning) => guardWarnings.push(warning)); - const pump = () => - new NodeTelemetryPump({ - runRoot, - row, - reporters: [guarded], - policy, - cursorPath, - maxDeliveryAttempts: 2, - retryDelayMs: 0 - }); - - const failed = await pump().drain(); - expect(reporter.attempts).toBe(2); - expect(failed.deliveredEvents).toBe(0); - expect(failed.warnings).toEqual( - expect.arrayContaining([expect.objectContaining({ code: "EVAL_TELEMETRY_DELIVERY_FAILED" })]) - ); - expect(guardWarnings).toEqual([]); - expect(loadTelemetryCursor(cursorPath)).toMatchObject({ byteOffset: 0, deliveredEventIds: [] }); - - const resumed = await pump().drain(); - expect(reporter.attempts).toBe(3); - expect(resumed).toMatchObject({ deliveredEvents: 1, warnings: [] }); - expect(reporter.envelopes()).toHaveLength(1); - expect(loadTelemetryCursor(cursorPath).byteOffset).toBeGreaterThan(0); - }); - - it("waits for complete journal lines before advancing the cursor", async () => { - const { runRoot, reporter, pump } = setup(); - writeRunFixture({ runRoot, events: [] }); - const eventsPath = path.join(runRoot, "events.jsonl"); - const record = JSON.stringify(completeEventRecord(nodeSyncedEvent("partial-line", T0, "running"))); - fs.writeFileSync(eventsPath, record.slice(0, 20), "utf8"); - await pump().drain(); - expect(reporter.envelopes()).toHaveLength(0); - fs.writeFileSync(eventsPath, `${record}\n`, "utf8"); - await pump().drain(); - expect(reporter.envelopes()).toHaveLength(1); - }); - - it("does not require an attempt for node states that do not produce a reporter transition", async () => { - const { runRoot, reporter, pump } = setup(); - writeRunFixture({ runRoot, events: [nodeSyncedEvent("pending", T0, "pending", 0)] }); - - await expect(pump().drain()).resolves.toMatchObject({ deliveredEvents: 0, warnings: [] }); - expect(reporter.envelopes()).toEqual([]); - }); - - it("rejects translated node transitions without a positive canonical payload attempt", async () => { - const { runRoot, cursorPath, reporter, pump } = setup(); - writeRunFixture({ - runRoot, - events: [ - { - event_id: eventId("missing-attempt"), - event_type: "node-synced", - timestamp: T0, - node_id: "setup-1", - status: "running", - payload: { workflow_run_id: "workflow-1", workflow_task_id: "node:setup-1" } - } - ] - }); - - await expect(pump().drain()).rejects.toMatchObject({ code: "EVAL_TELEMETRY_EVENT_UNUSABLE" }); - expect(reporter.envelopes()).toEqual([]); - expect(loadTelemetryCursor(cursorPath).byteOffset).toBe(0); - }); - - it("fails closed on schema-invalid complete records and records for another run", async () => { - const { runRoot, cursorPath, reporter, pump } = setup(); - writeRunFixture({ runRoot, events: [] }); - const eventsPath = path.join(runRoot, "events.jsonl"); - const canonical = completeEventRecord(nodeSyncedEvent("strict-row", T0, "running")); - - fs.writeFileSync(eventsPath, `${JSON.stringify({ ...canonical, schema_version: "1.0" })}\n`, "utf8"); - await expect(pump().drain()).rejects.toMatchObject({ code: "EVAL_TELEMETRY_JOURNAL_MALFORMED" }); - - fs.writeFileSync(eventsPath, `${JSON.stringify({ ...canonical, run_id: "another-run" })}\n`, "utf8"); - await expect(pump().drain()).rejects.toMatchObject({ code: "EVAL_TELEMETRY_JOURNAL_MALFORMED" }); - - expect(reporter.envelopes()).toEqual([]); - expect(loadTelemetryCursor(cursorPath).byteOffset).toBe(0); - }); - - it("rejects duplicate identities and out-of-order timestamps across the complete journal history", async () => { - const { runRoot, cursorPath, reporter, pump } = setup(); - const first = nodeSyncedEvent("history-first", T1, "running"); - const second = nodeSyncedEvent("history-second", T0, "succeeded"); - writeRunFixture({ runRoot, events: [first, second] }); - - await expect(pump().drain()).rejects.toMatchObject({ code: "EVAL_TELEMETRY_JOURNAL_MALFORMED" }); - - writeRunFixture({ runRoot, events: [first, first] }); - await expect(pump().drain()).rejects.toMatchObject({ code: "EVAL_TELEMETRY_JOURNAL_MALFORMED" }); - - expect(reporter.envelopes()).toEqual([]); - expect(loadTelemetryCursor(cursorPath).byteOffset).toBe(0); - }); - - it("synthesizes rate-limited heartbeats for running nodes", async () => { - const { runRoot, reporter, pump } = setup(); - writeRunFixture({ - runRoot, - events: [], - state: currentRunState({ - runId: "run-1", - status: "running", - nodes: { - "strategies-1": { - status: "running", - started_at: T0, - finished_at: undefined, - wait_since: T0, - wait_reason: "active", - next_eligible_action: "task-complete" - }, - "strategies-2": { - status: "running", - retry_count: 2, - started_at: T0, - finished_at: undefined, - wait_since: T0, - wait_reason: "active", - next_eligible_action: "task-complete" - }, - "setup-1": {} - }, - overrides: { created_at: T0, started_at: T0, last_transition_at: T0 } - }), - graph: currentPlannedGraph(["strategies-1", "strategies-2", "setup-1"], undefined) - }); - await pump().drain(); - const heartbeats = reporter.envelopes().filter((envelope) => envelope.event.type === "node-heartbeat"); - expect(heartbeats).toHaveLength(2); - expect(heartbeats[0]?.event).toMatchObject({ type: "node-heartbeat", status: "running", activeSeconds: 600 }); - expect(heartbeats[1]?.event).toMatchObject({ type: "node-heartbeat", status: "retrying" }); - - // Draining again within the heartbeat interval must not emit more heartbeats. - await pump().drain(); - expect(reporter.envelopes().filter((envelope) => envelope.event.type === "node-heartbeat")).toHaveLength(2); - }); - - it("fails closed on present unreadable journals and can resume after the journal is restored", async () => { - const { runRoot, reporter, pump } = setup(); - writeRunFixture({ runRoot, events: [] }); - const eventsPath = path.join(runRoot, "events.jsonl"); - fs.rmSync(eventsPath); - fs.mkdirSync(eventsPath); - await expect(pump().drain()).rejects.toMatchObject({ code: "EVAL_TELEMETRY_JOURNAL_UNREADABLE" }); - expect(reporter.envelopes()).toHaveLength(0); - - // Once the journal is healthy again the pump resumes from its cursor. - fs.rmdirSync(eventsPath); - fs.writeFileSync( - eventsPath, - `${JSON.stringify(completeEventRecord(nodeSyncedEvent("restored", T0, "running")))}\n`, - "utf8" - ); - await pump().drain(); - expect(reporter.envelopes()).toHaveLength(1); - }); - - it("degrades reporter failures to warnings and keeps draining", async () => { - const { runRoot, reporter, pump } = setup(); - reporter.failOn = new Set(["onNodeEvent"]); - writeRunFixture({ - runRoot, - events: [nodeSyncedEvent("reporter-failure", T0, "running")] - }); - const result = await pump().drain(); - expect(result.warnings.some((warning) => warning.code === "EVAL_TELEMETRY_DELIVERY_FAILED")).toBe(true); - expect(result.warnings.every((warning) => warning.severity === "warning")).toBe(true); - }); -}); diff --git a/packages/evals/test/reporters.test.ts b/packages/evals/test/reporters.test.ts index bb983ccda..02a7213eb 100644 --- a/packages/evals/test/reporters.test.ts +++ b/packages/evals/test/reporters.test.ts @@ -1,14 +1,11 @@ import { describe, expect, it } from "vitest"; -import { graphFromPlannedGraph } from "../src/reporter.js"; -import { boundedResponseText } from "../src/reporters/http.js"; import { EVAL_PROVIDER_NONE, KNOWN_EVAL_PROVIDERS, resolveEvalProvider, resolveEvalSuitePath } from "../src/reporters/index.js"; -import { currentPlannedGraph } from "./helpers.js"; const LEGACY_EVAL_CONFIG = { provider: "braintrust", @@ -85,69 +82,3 @@ describe("local provider resolution", () => { expect(resolveEvalSuitePath({ env: {} })).toBe(".ultrafuzz/evals/bug-finding.yml"); }); }); - -describe("provider transport", () => { - it("rejects provider responses larger than the transport limit", async () => { - const response = new Response("oversized", { - headers: { "content-length": String(1024 * 1024 + 1) } - }); - await expect(boundedResponseText(response, "provider", "EVAL_TEST_RESPONSE_TOO_LARGE")).rejects.toMatchObject({ - code: "EVAL_TEST_RESPONSE_TOO_LARGE" - }); - }); -}); - -describe("graphFromPlannedGraph", () => { - it("maps planned nodes into the provider row graph with group inference", () => { - const planned = currentPlannedGraph(["setup-1-a1", "strategies-fuzz-a1", "mystery"], undefined); - planned.groups = { setup: {}, strategies: {} }; - planned.nodes[0]!.logical_id = "setup-1"; - planned.nodes[0]!.model_fanout[0]!.model_profile_id = "eval-runner"; - planned.nodes[0]!.model_fanout[0]!.model_name = "gpt-5.4-mini"; - planned.nodes[1]!.depends_on = ["setup-1-a1"]; - Object.assign(planned.nodes[2]!, { - kind: "reference", - prompt_path: "", - model_fanout: [], - reference: "monad-developers/ultrafuzz", - reference_revision: { - provider: "github", - repo: "monad-developers/ultrafuzz", - commit: "a".repeat(40), - paths: ["README.md"] - } - }); - - const graph = graphFromPlannedGraph(planned, "row-1"); - expect(graph.rowId).toBe("row-1"); - expect(graph.nodes).toHaveLength(3); - expect(graph.nodes[0]).toMatchObject({ - id: "setup-1-a1", - logicalId: "setup-1", - group: "setup", - modelProfileId: "eval-runner", - model: "gpt-5.4-mini", - loopIndex: 0 - }); - expect(graph.nodes[1]).toMatchObject({ group: "strategies", dependsOn: ["setup-1-a1"] }); - expect(graph.nodes[2]).toMatchObject({ group: "default", kind: "reference" }); - }); - - it("uses an empty graph only when graph.json is absent", () => { - expect(graphFromPlannedGraph(undefined, "row-1")).toEqual({ rowId: "row-1", nodes: [] }); - }); - - it("rejects every malformed, schema-invalid, or semantically invalid present graph", () => { - expect(() => graphFromPlannedGraph({ nodes: "nope" }, "row-1")).toThrow("planned graph is schema-invalid"); - - const missingLogicalId = currentPlannedGraph(["setup-1"], undefined); - Reflect.deleteProperty(missingLogicalId.nodes[0]!, "logical_id"); - expect(() => graphFromPlannedGraph(missingLogicalId, "row-1")).toThrow("planned graph is schema-invalid"); - - const unknownDependency = currentPlannedGraph(["setup-1"], undefined); - unknownDependency.nodes[0]!.depends_on = ["missing-node"]; - expect(() => graphFromPlannedGraph(unknownDependency, "row-1")).toThrow( - 'planned graph node "setup-1" depends on unknown node "missing-node"' - ); - }); -}); diff --git a/packages/evals/test/runner-publish.test.ts b/packages/evals/test/runner-publish.test.ts index d7cef7872..3a43b33ea 100644 --- a/packages/evals/test/runner-publish.test.ts +++ b/packages/evals/test/runner-publish.test.ts @@ -3,7 +3,12 @@ import { mkdtempSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; -import { createNodeAttemptLedgerEntry } from "@ultrafuzz/artifacts"; +import { + EVENT_SCHEMA_VERSION, + assertEventRecord, + createNodeAttemptLedgerEntry, + readRunState +} from "@ultrafuzz/artifacts"; import { initProject } from "@ultrafuzz/runtime"; import { describe, expect, it, vi } from "vitest"; @@ -12,7 +17,6 @@ import { appendEvalRunRecord, readEvalMatrix, readEvalRunRecords } from "../src/ import { launchEvalRow, runEvalSuite, watchEvalRow } from "../src/runner.js"; import { boundedEvalWorkflowRunId } from "../src/utils.js"; import { - RecordingReporter, currentEvalRunRecord, currentPlannedGraph, currentRunState, @@ -964,6 +968,42 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. expect(readEvalRunRecords(path.join(result.eval_run_root, "runs.jsonl"))).toHaveLength(2); }); + it("publishes the run summary for a watched row whose skipped node synced without an attempt", async () => { + const { project, groundTruthRoot, suitePath } = initializationFixture("ufz-evals-run-attemptless-skip-"); + const row = testRow(testSuite(groundTruthRoot)); + const runId = expectedRowRunId("eval-attemptless-skip", row); + const runRoot = path.join(path.dirname(project), "target", ".ultrafuzz", "runs", runId); + terminalRunFixture(runRoot, "succeeded", runId); + // The event contract makes `payload.attempt` optional; sync omits it when the runner reported none. + const skipped = assertEventRecord({ + schema_version: EVENT_SCHEMA_VERSION, + run_id: runId, + event_id: `evt-${"4".repeat(24)}`, + event_type: "node-synced", + timestamp: T1, + node_id: "final-report", + status: "skipped", + payload: { workflow_run_id: "workflow-1", workflow_task_id: "node:final-report" } + }); + fs.appendFileSync(path.join(runRoot, "events.jsonl"), `${JSON.stringify(skipped)}\n`, "utf8"); + + const result = await runEvalSuite({ + projectRoot: project, + suitePath, + evalRunId: "eval-attemptless-skip", + groundTruthRoot, + provider: "none", + launcher: async () => ({ ok: true, runId, runRoot, workflowIds: ["workflow-1"], diagnostics: [] }) + }); + + expect(result).toMatchObject({ launched: 1, failed: 0, incomplete: 0 }); + expect(result.records[0]).toMatchObject({ final_status: "succeeded", workflow: { terminal: true } }); + const summary = JSON.parse(fs.readFileSync(path.join(result.eval_run_root, "run-summary.json"), "utf8")) as { + incomplete: number; + }; + expect(summary.incomplete).toBe(0); + }); + it("counts a watched row as incomplete when it misses the watch deadline", async () => { const base = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ufz-evals-run-watch-timeout-")); const project = path.join(base, "project"); @@ -1104,13 +1144,12 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. }); }); - it("watches a row to terminal state, draining telemetry after each sync tick", async () => { + it("records a row that is already terminal without synchronizing it", async () => { const base = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ufz-evals-watch-")); const suite = testSuite(path.join(base, "gt")); const row = testRow(suite); const runRoot = path.join(base, "target", ".ultrafuzz", "runs", "run-1"); terminalRunFixture(runRoot); - const reporter = new RecordingReporter(); let syncCalls = 0; const evalRunRoot = path.join(base, "eval-run"); @@ -1118,7 +1157,6 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, row, record: materializeLaunchedJournal(row, runRoot, evalRunRoot), - reporters: [reporter], evalRunRoot, sync: async () => { syncCalls += 1; @@ -1129,87 +1167,7 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. expect(watched.record.final_status).toBe("succeeded"); expect(watched.record.workflow).toMatchObject({ status: "succeeded", terminal: true, finished_at: T1 }); expect(readEvalRunRecords(path.join(base, "eval-run", "runs.jsonl")).at(-1)?.workflow?.finished_at).toBe(T1); - const methods = reporter.calls.map((call) => call.method); - expect(methods[0]).toBe("onRowStart"); - expect(methods[methods.length - 1]).toBe("onRowFinish"); - expect(reporter.envelopes().map((envelope) => envelope.event.type)).toEqual([ - "node-started", - "node-artifacts", - "node-finished" - ]); - const rowStartGraph = reporter.calls[0]?.args[1] as { nodes: Array<{ group: string }> }; - expect(rowStartGraph.nodes[0]?.group).toBe("setup"); - const finish = reporter.calls[reporter.calls.length - 1]?.args[1] as { status: string; startedAt?: string }; - expect(finish).toMatchObject({ status: "succeeded", startedAt: T0 }); - // Terminal state on entry: no polling sleep loops required. expect(syncCalls).toBe(0); - // Cursor persisted under the eval run root for crash-safe resume. - expect(fs.existsSync(path.join(base, "eval-run", "telemetry", `${row.id}.cursor.json`))).toBe(true); - }); - - it("rejects a present graph that would require telemetry repair", async () => { - const base = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ufz-evals-watch-invalid-graph-")); - const suite = testSuite(path.join(base, "gt")); - const row = testRow(suite); - const runRoot = path.join(base, "target", ".ultrafuzz", "runs", "run-1"); - terminalRunFixture(runRoot); - const invalidGraph = currentPlannedGraph(["setup-1"], undefined); - Reflect.deleteProperty(invalidGraph.nodes[0]!, "logical_id"); - fs.writeFileSync(path.join(runRoot, "graph.json"), JSON.stringify(invalidGraph), "utf8"); - const reporter = new RecordingReporter(); - const evalRunRoot = path.join(base, "eval-run"); - - await expect( - watchEvalRow({ - plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, - row, - record: materializeLaunchedJournal(row, runRoot, evalRunRoot), - reporters: [reporter], - evalRunRoot, - sync: async () => undefined, - pollIntervalMs: 1 - }) - ).rejects.toThrow("planned graph is schema-invalid"); - expect(reporter.calls).toEqual([]); - }); - - it("rejects a graph that disappears after the initial existence inspection", async () => { - const base = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ufz-evals-watch-graph-race-")); - const suite = testSuite(path.join(base, "gt")); - const row = testRow(suite); - const runRoot = path.join(base, "target", ".ultrafuzz", "runs", "run-1"); - terminalRunFixture(runRoot); - const graphPath = path.join(runRoot, "graph.json"); - const reporter = new RecordingReporter(); - const evalRunRoot = path.join(base, "eval-run"); - const record = materializeLaunchedJournal(row, runRoot, evalRunRoot); - const originalLstat = fs.lstatSync.bind(fs); - let removed = false; - const lstat = vi.spyOn(fs, "lstatSync").mockImplementation(((candidate: fs.PathLike) => { - const observed = originalLstat(candidate); - if (!removed && path.resolve(String(candidate)) === graphPath) { - removed = true; - fs.unlinkSync(graphPath); - } - return observed; - }) as typeof fs.lstatSync); - try { - await expect( - watchEvalRow({ - plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, - row, - record, - reporters: [reporter], - evalRunRoot, - sync: async () => undefined, - pollIntervalMs: 1 - }) - ).rejects.toThrow("cannot open regular file"); - } finally { - lstat.mockRestore(); - } - expect(removed).toBe(true); - expect(reporter.calls).toEqual([]); }); it("rejects a present dangling run state instead of treating the row as merely launched", async () => { @@ -1221,7 +1179,6 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. const statePath = path.join(runRoot, "state.json"); fs.unlinkSync(statePath); fs.symlinkSync("missing-state.json", statePath); - const reporter = new RecordingReporter(); const evalRunRoot = path.join(base, "eval-run"); const record = materializeLaunchedJournal(row, runRoot, evalRunRoot); @@ -1230,13 +1187,11 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, row, record, - reporters: [reporter], evalRunRoot, sync: async () => undefined, pollIntervalMs: 1 }) ).rejects.toThrow(); - expect(reporter.calls.map((call) => call.method)).toEqual(["onRowStart"]); expect(readEvalRunRecords(path.join(evalRunRoot, "runs.jsonl"))).toEqual([record]); }); @@ -1256,7 +1211,6 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. }), graph: currentPlannedGraph([], undefined) }); - const reporter = new RecordingReporter(); const evalRunRoot = path.join(base, "eval-run"); const record = materializeLaunchedJournal(row, runRoot, evalRunRoot); let swapped = false; @@ -1266,7 +1220,6 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, row, record, - reporters: [reporter], evalRunRoot, sync: async () => { swapped = true; @@ -1278,10 +1231,9 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. }, pollIntervalMs: 1 }) - ).rejects.toThrow('run state belongs to "foreign-run", expected "run-1"'); + ).rejects.toThrow('run state identity does not match run "run-1"'); expect(swapped).toBe(true); - expect(reporter.calls.map((call) => call.method)).toEqual(["onRowStart"]); expect(readEvalRunRecords(path.join(evalRunRoot, "runs.jsonl"))).toEqual([record]); }); @@ -1301,7 +1253,6 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. }), graph: currentPlannedGraph([], undefined) }); - const reporter = new RecordingReporter(); const evalRunRoot = path.join(base, "eval-run"); const record = materializeLaunchedJournal(row, runRoot, evalRunRoot); const sync = vi.fn(async () => undefined); @@ -1311,7 +1262,6 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, row, record, - reporters: [reporter], evalRunRoot, sync, pollIntervalMs: 1 @@ -1319,16 +1269,14 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. ).rejects.toThrow('run state identity does not match run "run-1"'); expect(sync).not.toHaveBeenCalled(); - expect(reporter.calls.map((call) => call.method)).toEqual(["onRowStart"]); expect(readEvalRunRecords(path.join(evalRunRoot, "runs.jsonl"))).toEqual([record]); }); - it("defers onRowStart until the detached subprocess writes graph.json", async () => { - const base = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ufz-evals-watch-race-")); + it("synchronizes a running row on every poll until its durable state is terminal", async () => { + const base = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ufz-evals-watch-poll-")); const suite = testSuite(path.join(base, "gt")); const row = testRow(suite); const runRoot = path.join(base, "target", ".ultrafuzz", "runs", "run-1"); - // Freshly launched run: state.json exists but DAG planning has not written graph.json yet. writeRunFixture({ runRoot, events: [], @@ -1339,7 +1287,6 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. overrides: { created_at: T0, started_at: T0, last_transition_at: T0 } }) }); - const reporter = new RecordingReporter(); let syncCalls = 0; const evalRunRoot = path.join(base, "eval-run"); @@ -1347,33 +1294,20 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, row, record: materializeLaunchedJournal(row, runRoot, evalRunRoot), - reporters: [reporter], evalRunRoot, sync: async () => { syncCalls += 1; - if (syncCalls === 2) { - // Second tick: planning finishes (graph.json appears) and the run completes. - terminalRunFixture(runRoot); - } + // The second synchronization observes the completed run. + if (syncCalls === 2) terminalRunFixture(runRoot); }, pollIntervalMs: 1 }); - expect(watched.record.final_status).toBe("succeeded"); expect(syncCalls).toBe(2); - const methods = reporter.calls.map((call) => call.method); - expect(methods[0]).toBe("onRowStart"); - expect(methods[methods.length - 1]).toBe("onRowFinish"); - // onRowStart waited for graph.json: reporters get the real node list, not an empty graph. - const rowStartGraph = reporter.calls[0]?.args[1] as { nodes: Array<{ id: string; group: string }> }; - expect(rowStartGraph.nodes).toHaveLength(2); - expect(rowStartGraph.nodes[0]).toMatchObject({ id: "setup-1", group: "setup" }); - // No node events were delivered before onRowStart, and all journal events still arrive. - expect(reporter.envelopes().map((envelope) => envelope.event.type)).toEqual([ - "node-started", - "node-artifacts", - "node-finished" - ]); + expect(watched.record).toMatchObject({ + final_status: "succeeded", + workflow: { status: "succeeded", terminal: true, finished_at: T1 } + }); }); it("records a typed incomplete outcome when the watch deadline expires", async () => { @@ -1398,7 +1332,6 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, row, record: materializeLaunchedJournal(row, runRoot, evalRunRoot), - reporters: [], evalRunRoot, sync: async () => undefined, pollIntervalMs: 1, @@ -1414,6 +1347,48 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. expect(watched.record.diagnostics.map((diagnostic) => diagnostic.code)).toContain("EVAL_ROW_WATCH_TIMEOUT"); }); + it("records the terminal status another writer reached while the watch slept past its deadline", async () => { + const base = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ufz-evals-watch-final-sleep-")); + const suite = testSuite(path.join(base, "gt")); + const row = testRow(suite); + const runRoot = path.join(base, "target", ".ultrafuzz", "runs", "run-1"); + writeRunFixture({ + runRoot, + events: [], + state: currentRunState({ + runId: "run-1", + status: "running", + nodes: {}, + overrides: { created_at: T0, started_at: T0, last_transition_at: T0 } + }) + }); + // Compile the state validator first so the one poll starts well inside the 1 s deadline. + readRunState(path.join(runRoot, "state.json")); + const evalRunRoot = path.join(base, "eval-run"); + let syncCalls = 0; + + const watched = await watchEvalRow({ + plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, + row, + record: materializeLaunchedJournal(row, runRoot, evalRunRoot), + evalRunRoot, + sync: async () => { + syncCalls += 1; + // Another syncer (`ultrafuzz status`, the dashboard) finishes the run during the 1.5 s sleep. + setTimeout(() => terminalRunFixture(runRoot), 300); + }, + pollIntervalMs: 1_500, + timeoutSeconds: 1 + }); + + expect(syncCalls).toBe(1); + expect(watched.record).toMatchObject({ + final_status: "succeeded", + workflow: { status: "succeeded", terminal: true } + }); + expect(watched.diagnostics.map((diagnostic) => diagnostic.code)).not.toContain("EVAL_ROW_WATCH_TIMEOUT"); + }, 20_000); + it("coalesces and persists workflow synchronization failures", async () => { const base = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ufz-evals-watch-sync-failure-")); const suite = testSuite(path.join(base, "gt")); @@ -1436,7 +1411,6 @@ Write a neutral fixture message to {{artifact_path}}/fixture.md. plan: { suite_path: "suite.yml", project_root: base, suite, matrix: [row] }, row, record: materializeLaunchedJournal(row, runRoot, evalRunRoot), - reporters: [], evalRunRoot, sync: async () => { syncCalls += 1; diff --git a/packages/evals/test/scoring.test.ts b/packages/evals/test/scoring.test.ts index 3e6587a46..eb3f40ed3 100644 --- a/packages/evals/test/scoring.test.ts +++ b/packages/evals/test/scoring.test.ts @@ -1481,11 +1481,7 @@ describe("deterministic scorer math", () => { headers: init?.headers as Record, ...(init?.redirect !== undefined ? { redirect: init.redirect } : {}) }); - return { - ok: true, - status: 200, - text: async () => JSON.stringify({ choices: [{ message: { content: responseContent } }] }) - }; + return new Response(JSON.stringify({ choices: [{ message: { content: responseContent } }] }), { status: 200 }); }) as unknown as typeof fetch; const judge = gatewayLlmJudge( { @@ -1704,6 +1700,18 @@ describe("deterministic scorer math", () => { expect(gateway.sleeps).toEqual([]); }); + it("does not retry a judge response larger than 1 MiB", async () => { + const gateway = scriptedGatewayJudge([ + async () => new Response("oversized", { headers: { "content-length": String(1024 * 1024 + 1) } }) + ]); + + await expect(scoreWithSingleJudge(gateway.judge)).rejects.toMatchObject({ + code: "EVAL_LLM_JUDGE_RESPONSE_TOO_LARGE", + details: { attempts: 1 } + }); + expect(gateway.bodies).toHaveLength(1); + }); + it("retries a 429 gateway response once it succeeds", async () => { const gateway = scriptedGatewayJudge([ async () => new Response("rate limited", { status: 429 }), diff --git a/packages/evmbench/package.json b/packages/evmbench/package.json index 967cc11e3..9e80ee483 100644 --- a/packages/evmbench/package.json +++ b/packages/evmbench/package.json @@ -19,7 +19,7 @@ "benchmark": "pnpm run build && node dist/cli.js", "build": "tsc -p tsconfig.build.json && rm -rf dist/schema && cp -R src/schema dist/schema", "lock": "pnpm run build && node dist/update-lock.js", - "test": "vitest run", + "test": "vitest run --testTimeout=30000", "typecheck": "tsc -p tsconfig.json --noEmit --pretty false" }, "dependencies": { diff --git a/packages/evmbench/src/schema-registry.ts b/packages/evmbench/src/schema-registry.ts index 7374ed7b9..2b370e618 100644 --- a/packages/evmbench/src/schema-registry.ts +++ b/packages/evmbench/src/schema-registry.ts @@ -142,12 +142,6 @@ export function evmbenchSchemaBundleDigest(): string { return schemaRegistryBundleDigest(evmbenchSchemaRegistry()); } -export function evmbenchSchemaPath(schemaId: string): string { - const entry = evmbenchSchemaRegistry().find((candidate) => candidate.id === schemaId); - if (entry === undefined) throw new Error(`unregistered EVMBench schema: ${schemaId}`); - return path.join(evmbenchSchemaDirectory(), entry.filename); -} - export function validateEvmbenchJsonSchema(schemaId: string, value: unknown): JsonSchemaValidationResult { const registry = evmbenchSchemaRegistry(); const entry = registry.find((candidate) => candidate.id === schemaId); diff --git a/packages/modal/package.json b/packages/modal/package.json index 350d48ae2..735c26160 100644 --- a/packages/modal/package.json +++ b/packages/modal/package.json @@ -22,7 +22,7 @@ "scripts": { "build": "tsc -p tsconfig.build.json && rm -rf dist/schema && cp -R schema dist/schema && node scripts/verify-schema-registry.mjs", "smoke": "pnpm run build && node dist/cli.js smoke", - "test": "vitest run", + "test": "vitest run --testTimeout=30000", "typecheck": "pnpm --filter @ultrafuzz/modal^... build && tsc -p tsconfig.json --noEmit --pretty false" }, "dependencies": { diff --git a/packages/modal/src/config.ts b/packages/modal/src/config.ts index cd055d46a..dacb87aef 100644 --- a/packages/modal/src/config.ts +++ b/packages/modal/src/config.ts @@ -3,8 +3,6 @@ import { resolveConfig } from "@ultrafuzz/config"; import { MODAL_BENCHMARK_CONFIG_SCHEMA_ID, type StrictModalBenchmarkConfigDocument } from "./modal-contracts.js"; import { assertModalDocumentValue, readModalDocument } from "./modal-documents.js"; -import { validateModalJsonSchema } from "./modal-schema-registry.js"; -import { modalBenchmarkConfigZodSchema } from "./benchmark-config-zod.js"; import { MODAL_PUBLIC_FULL_SANDBOX_TIMEOUT_MS, MODAL_PUBLIC_SANDBOX_TIMEOUT_MS, @@ -62,13 +60,6 @@ export function fingerprintModalModel(model: ModalModelSpec): string { .digest("hex"); } -export function modalBenchmarkConfigValidatorsAgree(value: unknown): boolean { - return ( - validateModalJsonSchema(MODAL_BENCHMARK_CONFIG_SCHEMA_ID, value).ok === - modalBenchmarkConfigZodSchema.safeParse(value).success - ); -} - /** Refuse an execution envelope that cannot contain even one complete campaign. * This is separate from document validation: current-schema configs remain inspectable. */ diff --git a/packages/modal/src/defaults.ts b/packages/modal/src/defaults.ts index fa9e77462..183820094 100644 --- a/packages/modal/src/defaults.ts +++ b/packages/modal/src/defaults.ts @@ -27,7 +27,6 @@ export const EVAL_WATCH_TIMEOUT_SECONDS = (MODAL_SANDBOX_TIMEOUT_MS - EVAL_POST_ export const MODAL_PRE_MODEL_RETRY_LIMIT = 3; export const MODAL_PRE_MODEL_RETRY_BASE_DELAY_MS = 1_000; export const MODAL_PRE_MODEL_RETRY_MAX_DELAY_MS = 4_000; -export const DEFAULT_NODE_TIMEOUT_SECONDS = 2 * 60 * 60; export const DEFAULT_MODAL_APP = "ultrafuzz-evals"; export const DEFAULT_MODAL_IMAGE = "ultrafuzz-security-runner:latest"; export const MODAL_BENCHMARK_SANDBOX_RESOURCES = { @@ -50,70 +49,3 @@ export interface ModalModelSpec { reasoning: string; auth_mode: ModelAuthMode; } - -export const DEFAULT_BENCHMARK_MODELS: readonly ModalModelSpec[] = [ - { - slug: "gpt-5-5", - model: "gpt-5.5", - provider: "openai", - agent: "CodexAgent", - reasoning: "xhigh", - auth_mode: "subscription" - }, - { - slug: "gpt-5-6-sol", - model: "gpt-5.6-sol", - provider: "openai", - agent: "CodexAgent", - reasoning: "xhigh", - auth_mode: "subscription" - }, - { - slug: "gpt-5-6-terra", - model: "gpt-5.6-terra", - provider: "openai", - agent: "CodexAgent", - reasoning: "xhigh", - auth_mode: "subscription" - }, - { - slug: "gpt-5-6-luna", - model: "gpt-5.6-luna", - provider: "openai", - agent: "CodexAgent", - reasoning: "xhigh", - auth_mode: "subscription" - }, - { - slug: "claude-fable-5", - model: "claude-fable-5", - provider: "anthropic", - agent: "ClaudeAgent", - reasoning: "max", - auth_mode: "subscription" - }, - { - slug: "claude-opus-4-8", - model: "claude-opus-4-8", - provider: "anthropic", - agent: "ClaudeAgent", - reasoning: "max", - auth_mode: "subscription" - }, - { - slug: "kimi-k3", - model: "kimi-k3", - provider: "kimi", - agent: "KimiAgent", - reasoning: "max", - auth_mode: "subscription" - }, - { - slug: "deepseek-v4-pro", - model: "deepseek-v4-pro", - provider: "deepseek", - agent: "DeepSeekAgent", - reasoning: "max", - auth_mode: "api-key" - } -] as const; diff --git a/packages/modal/src/layout.ts b/packages/modal/src/layout.ts index 87088dd8c..e337fca05 100644 --- a/packages/modal/src/layout.ts +++ b/packages/modal/src/layout.ts @@ -14,10 +14,6 @@ export function persistentDataRoot(runId: string, slug: string): string { return path.posix.join("/data", runId, slug); } -export function persistentWorkspaceRoot(runId: string, slug: string): string { - return path.posix.join(persistentDataRoot(runId, slug), "workspace"); -} - export function modalVolumeName(runId: string, slug: string): string { const identity = `${runId}\0${slug}`; const suffix = createHash("sha256").update(identity).digest("hex").slice(0, 12); diff --git a/packages/modal/src/modal-documents.ts b/packages/modal/src/modal-documents.ts index 6d7444434..544d2f7ae 100644 --- a/packages/modal/src/modal-documents.ts +++ b/packages/modal/src/modal-documents.ts @@ -6,7 +6,6 @@ import path from "node:path"; import { assertNoSymlinkComponents, assertPathInside, - DEFAULT_MAX_JSON_INSTANCE_BYTES, parseStrictJsonBytes, readRegularFileSnapshot, validateRegisteredJsonBytesSync, @@ -24,8 +23,6 @@ import { } from "./modal-schema-registry.js"; import { assertModalDocumentSemantics } from "./modal-semantic-gates.js"; -export const MAX_MODAL_DOCUMENT_BYTES = DEFAULT_MAX_JSON_INSTANCE_BYTES; - export interface ModalDocumentSnapshot { readonly schema_id: SchemaId; readonly schema_sha256: string; diff --git a/packages/modal/src/public-worker.ts b/packages/modal/src/public-worker.ts index e45472acc..5664e21e7 100644 --- a/packages/modal/src/public-worker.ts +++ b/packages/modal/src/public-worker.ts @@ -9,8 +9,6 @@ import { adaptBenchmarkManifestToEvalSuite, benchmarkLaneConcurrency, benchmarkLaneSelectedTargetIds, - BENCHMARK_FULL_MAX_PARALLEL_RUNS, - BENCHMARK_SMOKE_MAX_PARALLEL_RUNS, boundedEvalId, evalSuiteInputDocument, evalRunRoot, @@ -84,13 +82,11 @@ import { OperationalDispositionError, runNamingUnhandledFailure } from "./termin const ULTRAFUZZ_ROOT = "/opt/ultrafuzz"; const BAKED_CANDIDATE_ARCHIVE = "/opt/ultrafuzz-source.tgz"; -export const PUBLIC_SMITHERS_SEED_ROOT = "/opt/ultrafuzz-smithers-seed"; +const PUBLIC_SMITHERS_SEED_ROOT = "/opt/ultrafuzz-smithers-seed"; const CLI = path.join(ULTRAFUZZ_ROOT, "packages/cli/dist/index.js"); const PUBLIC_BUNDLE_FILE = "public-results.json"; const PUBLIC_WORKSPACE_ROOT = "/tmp/ultrafuzz-public-workspace"; -export const PUBLIC_BENCHMARK_MAX_PARALLEL_EVAL_ROWS = BENCHMARK_SMOKE_MAX_PARALLEL_RUNS; -export const PUBLIC_FULL_BENCHMARK_MAX_PARALLEL_EVAL_ROWS = BENCHMARK_FULL_MAX_PARALLEL_RUNS; -export const PUBLIC_BENCHMARK_PREPARATION_PARALLELISM = 8; +const PUBLIC_BENCHMARK_PREPARATION_PARALLELISM = 8; export const PUBLIC_BENCHMARK_EVAL_CLEANUP_SECONDS = 5 * 60; export const PUBLIC_BENCHMARK_SCORE_PER_WAVE_TIMEOUT_SECONDS = 45 * 60; export const PUBLIC_BENCHMARK_REPORT_TIMEOUT_SECONDS = 5 * 60; @@ -99,12 +95,12 @@ export const PUBLIC_BENCHMARK_PREPARATION_TIMEOUT_SECONDS = 20 * 60; // whose runner is OpenRouter runs one eval row at a time and one workflow node inside it. // Every trusted policy re-derivation reads this constant, so the launch, cleanup, and // publication guardrails agree with the suite the Modal worker actually runs. -export const PUBLIC_BENCHMARK_OPENROUTER_MAX_PARALLEL = 1; +const PUBLIC_BENCHMARK_OPENROUTER_MAX_PARALLEL = 1; // The smoke graph has four sequential agent stages. Each stage may use both of // its 1,800-second attempts, so retain ten minutes beyond the four-hour // topology bound for workflow transitions and final synchronization. -export const PUBLIC_BENCHMARK_SMOKE_MAX_RUNTIME_SECONDS = 4 * 60 * 60 + 10 * 60; -export const PUBLIC_BENCHMARK_THREAT_MODEL_MAX_RUNTIME_SECONDS = 15_000; +const PUBLIC_BENCHMARK_SMOKE_MAX_RUNTIME_SECONDS = 4 * 60 * 60 + 10 * 60; +const PUBLIC_BENCHMARK_THREAT_MODEL_MAX_RUNTIME_SECONDS = 15_000; // A manual full row retains the packaged specialist timeouts, including the // 7,200-second invariant campaign. This is a bounded row execution budget, not // a guarantee that every topology node can consume its worst-case timeout. @@ -150,10 +146,8 @@ type PublicDiagnosticSecretValues = readonly string[] | (() => Promise { }); } -/** - * The strict form. Retained for tests and for any caller that has already established the run must be - * resumable; `worker.ts` deliberately uses the tolerant lookup above instead. - */ -export async function locateModalResumeWorkspace(workRoot: string): Promise { - const found = await findModalResumeWorkspace(workRoot); - if (found.kind === "not-started") throw new Error(found.reason); - return found.workspace; -} - export async function finalizeModalEvalRunRecord( workspace: ModalResumeWorkspace, state: ModalResumeRunState, diff --git a/packages/modal/src/runner.ts b/packages/modal/src/runner.ts index dcb9e797e..254e39877 100644 --- a/packages/modal/src/runner.ts +++ b/packages/modal/src/runner.ts @@ -2959,18 +2959,6 @@ export function modalSecurityToolchainCommands(): string[] { ]; } -export function createTrackedSourceArchive( - repoRoot: string, - archive = path.join(realpathSync(os.tmpdir()), `ultrafuzz-modal-source-${String(process.pid)}.tgz`) -): string { - const trackedFiles = execFileSync("git", ["ls-files", "-z"], { cwd: repoRoot }); - if (trackedFiles.length === 0) throw new Error(`no Git-tracked source files found under ${repoRoot}`); - execFileSync("tar", ["--null", "-czf", archive, "-C", repoRoot, "--files-from=-"], { - input: trackedFiles - }); - return archive; -} - /** * Create a self-contained, shallow Git checkout for the exact candidate HEAD. * The worker extracts this immutable archive instead of cloning the candidate diff --git a/packages/modal/src/smoke.ts b/packages/modal/src/smoke.ts index 4384b1963..900570297 100644 --- a/packages/modal/src/smoke.ts +++ b/packages/modal/src/smoke.ts @@ -4,7 +4,6 @@ import { remoteAuthDir, remoteAuthPath } from "./layout.js"; export const MODAL_SMOKE_RESULT_SCHEMA_VERSION = "ultrafuzz.modal.smoke-result.v1" as const; export const MODAL_SMOKE_ENTRY_PATH = "/opt/ultrafuzz/packages/modal/dist/smoke-worker.js"; export const MODAL_SMOKE_DATA_ROOT = "/data/ultrafuzz-modal-smoke"; -export const MODAL_SMOKE_STOP_PATH = `${MODAL_SMOKE_DATA_ROOT}/fresh-stop`; export type ModalSmokePhase = "fresh" | "resume"; export type ModalSmokeFailureStage = diff --git a/packages/modal/src/worker-diagnostics.ts b/packages/modal/src/worker-diagnostics.ts index 856b8180c..d61d30e28 100644 --- a/packages/modal/src/worker-diagnostics.ts +++ b/packages/modal/src/worker-diagnostics.ts @@ -4,7 +4,7 @@ import { redactSecretsInText, SENSITIVE_REDACTION_PLACEHOLDER } from "@ultrafuzz * Bytes of child stderr retained while the rest is discarded, so a non-zero exit can name its own reason * without persisting provider responses or benchmark contents. */ -export const WORKER_STDERR_TAIL_BYTES = 8_192; +const WORKER_STDERR_TAIL_BYTES = 8_192; /** * Byte bound on one diagnostic message. It is the collector's bound, not this module's: diff --git a/packages/modal/src/worker-result.ts b/packages/modal/src/worker-result.ts index 3e04dc1b7..0687874d2 100644 --- a/packages/modal/src/worker-result.ts +++ b/packages/modal/src/worker-result.ts @@ -24,22 +24,6 @@ import { export const WORKER_RESULT_SCHEMA_VERSION = "ultrafuzz.modal.worker-result.v2" as const; -export const WORKER_RESULT_ALLOWED_KEYS = [ - "schema_version", - "result_type", - "generation", - "launch_generation", - "attempt", - "model_work_started", - "counts", - "checkpoint", - "exit_category", - "runtime_ms", - "usage", - "pricing", - "diagnostic_code" -] as const; - export const WORKER_DIAGNOSTIC_CODES = [ "worker-live", "worker-finished", diff --git a/packages/modal/test/ci-config.test.ts b/packages/modal/test/ci-config.test.ts index 6df0d84eb..29c0c72ef 100644 --- a/packages/modal/test/ci-config.test.ts +++ b/packages/modal/test/ci-config.test.ts @@ -4,7 +4,6 @@ import os from "node:os"; import path from "node:path"; import { describe, expect, it } from "vitest"; -import { parse } from "yaml"; import { MODAL_PUBLIC_FULL_SANDBOX_TIMEOUT_MS } from "../src/defaults.js"; import { @@ -203,98 +202,6 @@ describe("public Modal benchmark configuration", () => { expect(publicBenchmarkMaxRuntimeSeconds("full")).toBe(15_000); expect(manifest.control_timeout_seconds).toBe(37_500); expect(manifest.control_timeout_seconds * 1000).toBeLessThan(MODAL_PUBLIC_FULL_SANDBOX_TIMEOUT_MS); - const controlWindowPath = path.join(output, "control-window.json"); - execFileSync( - process.execPath, - [ - path.join(workspace, "scripts/ci/modal-benchmark-control-window.mjs"), - "create", - path.join(output, "manifest.json"), - controlWindowPath, - "d".repeat(40), - "https://github.com/monad-developers/ultrafuzz", - "23456-1", - "full" - ], - { cwd: workspace } - ); - const controlWindow = JSON.parse(fs.readFileSync(controlWindowPath, "utf8")) as { - schema_version: string; - candidate_commit: string; - repository: string; - generation: string; - mode: string; - manifest_sha256: string; - control_timeout_seconds: number; - started_at_epoch_seconds: number; - deadline_at_epoch_seconds: number; - }; - expect(controlWindow).toEqual( - expect.objectContaining({ - schema_version: "ultrafuzz.modal.ci-control-window.v1", - candidate_commit: "d".repeat(40), - repository: "https://github.com/monad-developers/ultrafuzz", - generation: "23456-1", - mode: "full", - control_timeout_seconds: 37_500, - manifest_sha256: expect.stringMatching(/^[0-9a-f]{64}$/u) - }) - ); - expect(controlWindow.deadline_at_epoch_seconds - controlWindow.started_at_epoch_seconds).toBe(37_500); - const restoredDeadline = Number( - execFileSync( - process.execPath, - [ - path.join(workspace, "scripts/ci/modal-benchmark-control-window.mjs"), - "deadline", - path.join(output, "manifest.json"), - controlWindowPath, - "d".repeat(40), - "https://github.com/monad-developers/ultrafuzz", - "23456-1", - "full" - ], - { cwd: workspace, encoding: "utf8" } - ).trim() - ); - expect(restoredDeadline).toBe(controlWindow.deadline_at_epoch_seconds); - const wrongRepository = spawnSync( - process.execPath, - [ - path.join(workspace, "scripts/ci/modal-benchmark-control-window.mjs"), - "deadline", - path.join(output, "manifest.json"), - controlWindowPath, - "d".repeat(40), - "https://github.com/monad-developers/other", - "23456-1", - "full" - ], - { cwd: workspace, encoding: "utf8" } - ); - expect(wrongRepository.status).not.toBe(0); - expect(wrongRepository.stderr).toMatch(/manifest identity does not match/u); - const tamperedWindowPath = path.join(output, "control-window-tampered.json"); - fs.writeFileSync( - tamperedWindowPath, - `${JSON.stringify({ ...controlWindow, deadline_at_epoch_seconds: restoredDeadline + 1 })}\n` - ); - const tamperedDeadline = spawnSync( - process.execPath, - [ - path.join(workspace, "scripts/ci/modal-benchmark-control-window.mjs"), - "deadline", - path.join(output, "manifest.json"), - tamperedWindowPath, - "d".repeat(40), - "https://github.com/monad-developers/ultrafuzz", - "23456-1", - "full" - ], - { cwd: workspace, encoding: "utf8" } - ); - expect(tamperedDeadline.status).not.toBe(0); - expect(tamperedDeadline.stderr).toMatch(/deadline is invalid/u); expect(manifest.concurrency.max_parallel_eval_rows_per_sandbox).toBe(maxParallel); expect(manifest.concurrency.max_parallel_workflow_nodes_per_row).toBe( publicBenchmarkMaxParallelWorkflowNodes("full") @@ -622,191 +529,6 @@ describe("public Modal benchmark configuration", () => { } }); - it("runs the release validation lanes the policy script selects, on pull requests too", () => { - const workspace = path.resolve("../.."); - const workflow = parse(fs.readFileSync(path.join(workspace, ".github/workflows/ci.yml"), "utf8")) as { - on: { - push: { branches: string[] }; - pull_request: { types: string[] }; - workflow_dispatch: null; - }; - concurrency: { group: string; "cancel-in-progress": boolean }; - jobs: Record< - string, - { - name?: string; - if?: string; - needs?: string[]; - "timeout-minutes"?: number | string; - outputs?: Record; - strategy?: { - "fail-fast": boolean; - "max-parallel": number; - // #994 replaced the hard-coded lane table with a matrix expression - // expanded from the `release-validation-lanes` job output. - matrix: { include: string }; - }; - steps: Array<{ - name?: string; - id?: string; - if?: string; - run?: string; - uses?: string; - env?: Record; - with?: Record; - }>; - } - >; - }; - - expect(workflow.on.push.branches).toEqual(["main"]); - expect(workflow.on).not.toHaveProperty("merge_group"); - expect(workflow.on.workflow_dispatch).toBeNull(); - expect(workflow.on.pull_request.types).toEqual([ - "opened", - "synchronize", - "reopened", - "ready_for_review", - "converted_to_draft" - ]); - expect(workflow.concurrency.group).toContain("github.event.pull_request.number"); - expect(workflow.concurrency.group).toContain("github.ref"); - expect(workflow.concurrency["cancel-in-progress"]).toBe(true); - - const draftAndBuild = workflow.jobs["draft-and-build-gates"]; - expect(draftAndBuild?.name).toBe( - "${{ github.event_name == 'pull_request' && 'PR build and Node.js 24 runtime smoke' || 'Build gates' }}" - ); - // The PR-only runtime and CLI smoke suites have reached the old 15-minute - // ceiling while still making progress, so pin the bounded completion budget. - expect(draftAndBuild?.["timeout-minutes"]).toBe(30); - const steps = draftAndBuild?.steps ?? []; - const bunSetup = steps.find((step) => step.name === "Set up Bun"); - expect(bunSetup?.uses).toBe("oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6"); - expect(bunSetup?.with?.["bun-version"]).toBe("1.3.14"); - for (const name of [ - "Test CI policy scripts", - "Enforce production dependency advisory policy", - "Check formatting", - "Lint", - "Build" - ]) { - expect(steps.find((step) => step.name === name)?.if, `${name} must run for drafts`).toBeUndefined(); - } - expect(steps.find((step) => step.name === "Test CI policy scripts")?.run).toBe("pnpm -w test:ci-scripts"); - expect(steps.find((step) => step.name === "Enforce production dependency advisory policy")?.run).toBe( - "pnpm -w security:dependency-advisories" - ); - const runtimeSmoke = steps.find((step) => step.name === "Run PR runtime smoke tests"); - expect(runtimeSmoke?.if).toBe("github.event_name == 'pull_request'"); - expect(runtimeSmoke?.run).toBe("pnpm --filter @ultrafuzz/runtime test:pr-smoke:prebuilt"); - const cliStatusSmoke = steps.find((step) => step.name === "Run PR CLI status contract smoke tests"); - expect(cliStatusSmoke?.if).toBe("github.event_name == 'pull_request'"); - expect(cliStatusSmoke?.run).toBe("pnpm --filter @ultrafuzz/cli test:pr-smoke:prebuilt"); - - expect(workflow.jobs).not.toHaveProperty("pull-request-validation"); - - // The selection job is the workflow's only source of lanes, so pin the wiring - // end to end: the job publishes what the policy script prints, and the - // matrix expands exactly that output. - const laneSelection = workflow.jobs["release-validation-lanes"]; - expect(laneSelection?.outputs?.lanes).toBe("${{ steps.select.outputs.lanes }}"); - const selectStep = laneSelection?.steps.find((step) => step.name === "Select release validation lanes"); - expect(selectStep?.id).toBe("select"); - expect(selectStep?.env?.EVENT_NAME).toBe("${{ github.event_name }}"); - expect(selectStep?.run).toContain('scripts/ci/release-validation-lanes.mjs --event "$EVENT_NAME"'); - - const releaseValidation = workflow.jobs["release-validation"]; - expect(releaseValidation?.name).toBe("Full release validation (${{ matrix.description }})"); - // Regression guard for the outage this design exists to prevent. While this - // job carried `if: github.event_name != 'pull_request'`, every runtime test - // reported `skipping` on pull requests, so resume-path regressions merged - // with all checks green. - expect(releaseValidation?.if, "release validation must not be gated off pull requests").toBeUndefined(); - expect(releaseValidation?.needs).toEqual([ - "draft-and-build-gates", - "external-static-analysis", - "release-validation-lanes" - ]); - expect(releaseValidation?.strategy).toEqual({ - "fail-fast": false, - "max-parallel": 8, - matrix: { include: "${{ fromJSON(needs.release-validation-lanes.outputs.lanes) }}" } - }); - expect(releaseValidation?.["timeout-minutes"]).toBe("${{ matrix.timeout_minutes }}"); - expect(releaseValidation?.steps.find((step) => step.name === "Validate release lane")?.run).toContain("--gates"); - const modalDependentLaneBuild = releaseValidation?.steps.find( - (step) => step.name === "Build Modal-dependent lane dependencies" - ); - expect(modalDependentLaneBuild?.if).toBe("matrix.build_modal_dependencies == true"); - expect(modalDependentLaneBuild?.run).toBe("pnpm --filter @ultrafuzz/modal... build"); - const releaseReporterBuild = releaseValidation?.steps.find( - (step) => step.name === "Build release reporter dependencies" - ); - expect(releaseReporterBuild?.if).toBe("matrix.build_release_reporter == true"); - expect(releaseReporterBuild?.run).toBe("pnpm --filter @ultrafuzz/artifacts... build"); - // The split benchmark-history lane no longer shares a job with the `cli` gate, - // so it has to build the CLI closure that `benchmark:check:prebuilt` executes. - const cliLaneBuild = releaseValidation?.steps.find((step) => step.name === "Build CLI lane dependencies"); - expect(cliLaneBuild?.if).toBe("matrix.build_cli == true"); - expect(cliLaneBuild?.run).toBe("pnpm --filter @ultrafuzz/cli... build"); - // The lane table moved out of this file, so read it back the way the - // workflow does — by running the policy script for an integration event — - // rather than dropping the coverage this test used to carry. - const selection = execFileSync( - process.execPath, - [path.join(workspace, "scripts/ci/release-validation-lanes.mjs"), "--event", "push"], - { cwd: workspace, encoding: "utf8" } - ); - const pushLanes = JSON.parse(selection) as Array<{ lane: string; gates: string; timeout_minutes: number }>; - const laneGateIds = pushLanes.flatMap((entry) => entry.gates.split(",")); - expect(new Set(laneGateIds).size, "release validation lanes must not repeat a gate").toBe(laneGateIds.length); - expect(laneGateIds).toContain("cli"); - expect(laneGateIds).toContain("benchmark-history"); - expect(laneGateIds).toContain("workspace-typecheck"); - expect(pushLanes.map((entry) => entry.lane)).toContain("package-gates"); - expect(releaseValidation?.steps.find((step) => step.name === "Validate benchmark history charts")).toBeUndefined(); - const releaseGates = workflow.jobs["release-gates"]; - expect(releaseGates?.needs).toEqual([ - "draft-and-build-gates", - "external-static-analysis", - "release-validation-lanes", - "release-validation" - ]); - expect(releaseGates?.steps.find((step) => step.name === "Require the release validation lane selection")?.if).toBe( - "needs.release-validation-lanes.result != 'success'" - ); - for (const name of [ - "Check out repository", - "Set up pnpm", - "Set up Node.js", - "Install dependencies", - "Build release reporter dependencies", - "Download release validation lanes", - "Merge release validation report" - ]) { - expect(releaseGates?.steps.find((step) => step.name === name)?.if).toBe("github.event_name != 'pull_request'"); - } - expect(releaseGates?.steps.find((step) => step.name === "Install dependencies")?.run).toBe( - "pnpm install --frozen-lockfile" - ); - expect(releaseGates?.steps.find((step) => step.name === "Build release reporter dependencies")?.run).toBe( - "pnpm --filter @ultrafuzz/artifacts... build" - ); - expect(releaseGates?.steps.find((step) => step.name === "Merge release validation report")?.run).toContain( - "--merge-report-dir" - ); - // Formerly gated on `github.event_name != 'pull_request'`; #994 made the - // lanes required on pull requests, so the requirement must apply to every - // event. - expect(releaseGates?.steps.find((step) => step.name === "Require release validation lanes")?.if).toBe( - "always() && needs.release-validation.result != 'success'" - ); - expect(releaseGates?.steps.find((step) => step.name === "Upload release validation report")?.if).toBe( - "always() && github.event_name != 'pull_request'" - ); - }); - it("keeps Actions limited to CI without provider secrets or paid benchmark launches", () => { const workspace = path.resolve("../.."); const workflowRoot = path.join(workspace, ".github/workflows"); diff --git a/packages/modal/test/config.test.ts b/packages/modal/test/config.test.ts index 76cad1563..98b9f2481 100644 --- a/packages/modal/test/config.test.ts +++ b/packages/modal/test/config.test.ts @@ -13,11 +13,9 @@ import { MODAL_GIT_URL_PATTERN_SOURCE, MODAL_HTTPS_URL_PATTERN_SOURCE, modalBenchmarkConfigZodSchema, - modalBenchmarkConfigValidatorsAgree, parseModalBenchmarkConfig } from "../src/config.js"; import { - DEFAULT_BENCHMARK_MODELS, MODAL_BENCHMARK_SCHEMA_VERSION, MODAL_PUBLIC_FULL_SANDBOX_TIMEOUT_MS, MODAL_PUBLIC_SANDBOX_TIMEOUT_MS, @@ -27,10 +25,18 @@ import { MODAL_BENCHMARK_CONFIG_SCHEMA_ID } from "../src/modal-contracts.js"; import { ModalDocumentValidationError } from "../src/modal-documents.js"; import { modalBenchmarkConfigJsonSchema, validateModalJsonSchema } from "../src/modal-schema-registry.js"; import { ModalSemanticValidationError } from "../src/modal-semantic-gates.js"; +import { MODEL_SPEC_FIXTURES } from "./model-spec-fixtures.js"; + +function modalBenchmarkConfigValidatorsAgree(value: unknown): boolean { + return ( + validateModalJsonSchema(MODAL_BENCHMARK_CONFIG_SCHEMA_ID, value).ok === + modalBenchmarkConfigZodSchema.safeParse(value).success + ); +} function minimalConfig(): Record { return { - ...commonConfig("example-run", [...DEFAULT_BENCHMARK_MODELS]), + ...commonConfig("example-run", [...MODEL_SPEC_FIXTURES]), target: { repo: "https://example.invalid/target.git", ref: "0123456789abcdef" }, ground_truth: { repo: "https://example.invalid/ground-truth.git", @@ -60,7 +66,7 @@ function commonConfig(runId: string, models: unknown[]): Record } function minimalPublicConfig() { - const model = DEFAULT_BENCHMARK_MODELS[0]!; + const model = MODEL_SPEC_FIXTURES[0]; return { ...commonConfig("public-run", [model]), public_benchmark: { @@ -232,7 +238,7 @@ describe("Modal benchmark config", () => { it("bounds new public sandboxes without cutting off the accepted full-lane envelope", () => { const privateConfig = parseModalBenchmarkConfig(minimalConfig()); - const model = DEFAULT_BENCHMARK_MODELS[0]!; + const model = MODEL_SPEC_FIXTURES[0]; const publicConfig = parseModalBenchmarkConfig({ ...commonConfig("public-run", [model]), public_benchmark: { @@ -307,7 +313,7 @@ describe("Modal benchmark config", () => { label: "duplicate projected model slugs", value: { ...minimalConfig(), - models: [DEFAULT_BENCHMARK_MODELS[0], DEFAULT_BENCHMARK_MODELS[0]] + models: [MODEL_SPEC_FIXTURES[0], MODEL_SPEC_FIXTURES[0]] } }, { @@ -349,7 +355,7 @@ describe("Modal benchmark config", () => { expect(() => parseModalBenchmarkConfig({ ...minimalConfig(), - models: [{ ...DEFAULT_BENCHMARK_MODELS[0], agent: "ClaudeAgent" }] + models: [{ ...MODEL_SPEC_FIXTURES[0], agent: "ClaudeAgent" }] }) ).toThrow(); }); @@ -634,8 +640,8 @@ describe("Modal benchmark config", () => { const first = fingerprintModalConfigFile(file); fs.writeFileSync(file, `${JSON.stringify(minimalConfig(), null, 2)}\n`); expect(fingerprintModalConfigFile(file)).not.toBe(first); - expect(fingerprintModalModel(DEFAULT_BENCHMARK_MODELS[0]!)).not.toBe( - fingerprintModalModel({ ...DEFAULT_BENCHMARK_MODELS[0]!, reasoning: "different" }) + expect(fingerprintModalModel(MODEL_SPEC_FIXTURES[0])).not.toBe( + fingerprintModalModel({ ...MODEL_SPEC_FIXTURES[0], reasoning: "different" }) ); }); @@ -660,7 +666,7 @@ describe("Modal benchmark config", () => { file, `${JSON.stringify({ ...config, - models: [DEFAULT_BENCHMARK_MODELS[0], DEFAULT_BENCHMARK_MODELS[0]] + models: [MODEL_SPEC_FIXTURES[0], MODEL_SPEC_FIXTURES[0]] })}\n` ); expect(() => loadModalBenchmarkConfig(file)).toThrow(/trusted semantic gates/u); @@ -672,7 +678,7 @@ describe("Modal benchmark config", () => { valid, { ...valid, unexpected: true }, { ...valid, loops: "3" }, - { ...valid, models: [DEFAULT_BENCHMARK_MODELS[0], DEFAULT_BENCHMARK_MODELS[0]] }, + { ...valid, models: [MODEL_SPEC_FIXTURES[0], MODEL_SPEC_FIXTURES[0]] }, { ...valid, benchmark_execution: { excluded_node_ids: ["boundary-tests", "boundary-tests"] } diff --git a/packages/modal/test/layout.test.ts b/packages/modal/test/layout.test.ts index 7e28e57fe..c1ba0fe10 100644 --- a/packages/modal/test/layout.test.ts +++ b/packages/modal/test/layout.test.ts @@ -14,7 +14,6 @@ import { REMOTE_LAUNCH_READY_PATH, REMOTE_LINEAGE_PATH, modalVolumeName, - persistentWorkspaceRoot, remoteAuthPath, resolvePersistentRemoteRoot } from "../src/layout.js"; @@ -22,7 +21,6 @@ import { modalEvalRunCommand } from "../src/resume.js"; describe("Modal storage layout", () => { it("persists workspaces while keeping config and auth ephemeral", () => { - expect(persistentWorkspaceRoot("run-1", "model-1")).toBe("/data/run-1/model-1/workspace"); expect(REMOTE_CONFIG_PATH).toBe("/run/ultrafuzz-config/benchmark.json"); expect(REMOTE_LINEAGE_PATH).toBe("/run/ultrafuzz-config/lineage.json"); expect(REMOTE_LAUNCH_READY_PATH).toBe("/run/ultrafuzz-config/launch-ready"); diff --git a/packages/modal/test/model-spec-fixtures.ts b/packages/modal/test/model-spec-fixtures.ts new file mode 100644 index 000000000..70d725d78 --- /dev/null +++ b/packages/modal/test/model-spec-fixtures.ts @@ -0,0 +1,69 @@ +import type { ModalModelSpec } from "../src/defaults.js"; + +/** Model specs used as ordinary configuration fixtures; production has no default model list. */ +export const MODEL_SPEC_FIXTURES = [ + { + slug: "gpt-5-5", + model: "gpt-5.5", + provider: "openai", + agent: "CodexAgent", + reasoning: "xhigh", + auth_mode: "subscription" + }, + { + slug: "gpt-5-6-sol", + model: "gpt-5.6-sol", + provider: "openai", + agent: "CodexAgent", + reasoning: "xhigh", + auth_mode: "subscription" + }, + { + slug: "gpt-5-6-terra", + model: "gpt-5.6-terra", + provider: "openai", + agent: "CodexAgent", + reasoning: "xhigh", + auth_mode: "subscription" + }, + { + slug: "gpt-5-6-luna", + model: "gpt-5.6-luna", + provider: "openai", + agent: "CodexAgent", + reasoning: "xhigh", + auth_mode: "subscription" + }, + { + slug: "claude-fable-5", + model: "claude-fable-5", + provider: "anthropic", + agent: "ClaudeAgent", + reasoning: "max", + auth_mode: "subscription" + }, + { + slug: "claude-opus-4-8", + model: "claude-opus-4-8", + provider: "anthropic", + agent: "ClaudeAgent", + reasoning: "max", + auth_mode: "subscription" + }, + { + slug: "kimi-k3", + model: "kimi-k3", + provider: "kimi", + agent: "KimiAgent", + reasoning: "max", + auth_mode: "subscription" + }, + { + slug: "deepseek-v4-pro", + model: "deepseek-v4-pro", + provider: "deepseek", + agent: "DeepSeekAgent", + reasoning: "max", + auth_mode: "api-key" + } +] as const satisfies readonly ModalModelSpec[]; diff --git a/packages/modal/test/public-worker.test.ts b/packages/modal/test/public-worker.test.ts index 2a37dd3b5..0b797567d 100644 --- a/packages/modal/test/public-worker.test.ts +++ b/packages/modal/test/public-worker.test.ts @@ -43,11 +43,9 @@ import { MAX_PUBLIC_OPTIONAL_ROW_ARTIFACT_FILES, PUBLIC_OPTIONAL_ROW_ARTIFACTS, PUBLIC_BENCHMARK_EVAL_CLEANUP_SECONDS, - PUBLIC_BENCHMARK_MAX_PARALLEL_EVAL_ROWS, PUBLIC_BENCHMARK_PREPARATION_TIMEOUT_SECONDS, PUBLIC_BENCHMARK_REPORT_TIMEOUT_SECONDS, PUBLIC_BENCHMARK_SCORE_PER_WAVE_TIMEOUT_SECONDS, - PUBLIC_FULL_BENCHMARK_MAX_PARALLEL_EVAL_ROWS, PublicEvalDiagnosticsBuildError, PublicWorkerCommandInterruptedError, publicBenchmarkMaxParallelEvalRows, @@ -60,8 +58,8 @@ import { publicEvalRunId, preparePublicEvalSuite, publicCommandExitFailureCause, - publicEvalFailureDiagnosticLogPayload, publicEvalFailureDiagnosticLogPayloadFromRecords, + publicEvalFailureEnvelopeDiagnostics, publicEvalModelWorkEvidence, publicEvalCommandLeftFinalJournal, publicEvalRunErrorCanBePublished, @@ -1222,6 +1220,14 @@ it("continues after the eval command reports one publishable failed datapoint", ).toBe(false); }); +function publicEvalFailureDiagnosticLogPayload( + stdout: string, + forbiddenSecretValues: readonly string[] +): string | undefined { + const diagnostics = publicEvalFailureEnvelopeDiagnostics(stdout); + return diagnostics === undefined ? undefined : workerDiagnosticLogPayload(diagnostics, forbiddenSecretValues); +} + it("publishes only bounded redacted workflow-submission messages from eval JSON", () => { const secret = "sk-fixture-secret-value"; const payload = publicEvalFailureDiagnosticLogPayload( @@ -1857,7 +1863,7 @@ it("bounds public provider fan-out by mode", () => { ...smokeBaseSuite, run: { ...smokeBaseSuite.run, - max_parallel_runs: PUBLIC_BENCHMARK_MAX_PARALLEL_EVAL_ROWS, + max_parallel_runs: 3, max_parallel_targets: 4 } }); @@ -1877,7 +1883,7 @@ it("bounds public provider fan-out by mode", () => { expect(fullSuite.targets).toHaveLength(40); expect(fullSuite.run).toEqual({ ...fullBaseSuite.run, - max_parallel_runs: PUBLIC_FULL_BENCHMARK_MAX_PARALLEL_EVAL_ROWS, + max_parallel_runs: 20, max_parallel_targets: 8 }); expect(publicBenchmarkMaxParallelEvalRows("smoke")).toBe(3); @@ -2687,9 +2693,8 @@ function writeGenuineTaskFailureFixture(runRoot: string): void { it("retains threat-model, goal-plan and vulnerability-database artifacts per row when the run produced them", () => { // #183 requires the real generated documents to be retrievable. They cannot - // reach the bundle any other way: `reporting.artifacts.include` is consumed - // only by `uploadsForManifest`, which delivers to `this.input.reporters`, and - // the public worker runs `eval run --provider none` with an empty reporter list. + // reach the bundle any other way: nothing in the eval runner reads + // `reporting.artifacts.include`. const root = fs.mkdtempSync(path.join(process.env.TMPDIR ?? "/tmp", "ultrafuzz-public-worker-threat-")); const controlRoot = path.join(root, "control"); const evalRunId = "eval-threat-model"; diff --git a/packages/modal/test/resume.test.ts b/packages/modal/test/resume.test.ts index ba908623c..61de64f8b 100644 --- a/packages/modal/test/resume.test.ts +++ b/packages/modal/test/resume.test.ts @@ -15,14 +15,14 @@ import { import { findModalResumeWorkspace, - locateModalResumeWorkspace, modalDurableResumeCommand, modalDurableRunAdvanced, modalDurableRunNeedsResume, modalEvalRunCommand, NonResumableTerminalRunError, finalizeModalEvalRunRecord, - readModalDurableRunState + readModalDurableRunState, + type ModalResumeWorkspace } from "../src/resume.js"; import { currentRunState } from "./current-artifact-fixtures.js"; @@ -213,6 +213,12 @@ function fixture() { return { workRoot, target, control, evalRunId, evalDir }; } +async function resumableWorkspace(workRoot: string): Promise { + const found = await findModalResumeWorkspace(workRoot); + if (found.kind === "not-started") throw new Error(found.reason); + return found.workspace; +} + describe("Modal durable evaluation resume", () => { it("uses durable resume without any node reset path", () => { expect(modalDurableResumeCommand("/opt/tool/cli.js", "durable-run-one", "/workspace/target")).toEqual([ @@ -334,21 +340,6 @@ describe("Modal durable evaluation resume", () => { expect(modalDurableRunAdvanced(before, { ...before, run_id: "durable-run-two", status: "running" })).toBe(false); }); - it("locates one exact linked durable run and rejects ambiguity", async () => { - const value = fixture(); - await expect(locateModalResumeWorkspace(value.workRoot)).resolves.toEqual({ - target: value.target, - control: value.control, - evalRunId: value.evalRunId, - productRunId: "durable-run-one" - }); - fs.appendFileSync( - path.join(value.evalDir, "runs.jsonl"), - `${JSON.stringify(currentEvalRunRecord(value.target, value.evalRunId, { rowId: "row-two", runId: "durable-run-two" }))}\n` - ); - await expect(locateModalResumeWorkspace(value.workRoot)).rejects.toThrow("exactly one linked durable run"); - }); - it("validates the exact current eval manifest and refuses historical, mismatched, or symlinked manifests", async () => { const value = fixture(); const manifestPath = path.join(value.evalDir, "eval.json"); @@ -662,7 +653,7 @@ describe("Modal durable evaluation resume", () => { it("finalizes succeeded and genuine task outcomes without resetting completed nodes", async () => { const value = fixture(); - const workspace = await locateModalResumeWorkspace(value.workRoot); + const workspace = await resumableWorkspace(value.workRoot); await finalizeModalEvalRunRecord( workspace, { run_id: "durable-run-one", status: "succeeded", started_at: T0, finished_at: T1 }, @@ -698,7 +689,7 @@ describe("Modal durable evaluation resume", () => { it("fails closed for operational terminal states and unrelated runs", async () => { const value = fixture(); - const workspace = await locateModalResumeWorkspace(value.workRoot); + const workspace = await resumableWorkspace(value.workRoot); const rejected = finalizeModalEvalRunRecord( workspace, { run_id: "durable-run-one", status: "failed" }, diff --git a/packages/modal/test/runner.test.ts b/packages/modal/test/runner.test.ts index d668dc906..3059a1874 100644 --- a/packages/modal/test/runner.test.ts +++ b/packages/modal/test/runner.test.ts @@ -57,7 +57,6 @@ import { createModalBenchmarkSandbox, createModalLaunchSandbox, createExactCandidateSourceArchive, - createTrackedSourceArchive, finishReservedModalLaunch, hasExactPublicDiagnosticCollectionConfig, isModalRecoveryResultComplete, @@ -515,25 +514,6 @@ describe("Modal image source staging", () => { expect(standaloneDockerfile).toContain("RUN command -v zstd"); }); - it("archives tracked files only", () => { - const root = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ultrafuzz-modal-archive-")); - execFileSync("git", ["init", "--quiet"], { cwd: root }); - fs.writeFileSync(path.join(root, ".gitignore"), ".private/\n", "utf8"); - fs.writeFileSync(path.join(root, "tracked.txt"), "tracked\n", "utf8"); - fs.writeFileSync(path.join(root, "untracked.txt"), "untracked\n", "utf8"); - fs.mkdirSync(path.join(root, ".private")); - fs.writeFileSync(path.join(root, ".private", "benchmark.json"), "private\n", "utf8"); - execFileSync("git", ["add", ".gitignore", "tracked.txt"], { cwd: root }); - const archive = path.join(root, "source.tgz"); - - createTrackedSourceArchive(root, archive); - const entries = execFileSync("tar", ["-tzf", archive], { encoding: "utf8" }).trim().split("\n"); - - expect(entries).toEqual(expect.arrayContaining([".gitignore", "tracked.txt"])); - expect(entries).not.toContain("untracked.txt"); - expect(entries).not.toContain(".private/benchmark.json"); - }); - it("bakes a clean shallow Git checkout at the exact candidate commit", () => { const root = mkdtempSync(path.join(fs.realpathSync(tmpdir()), "ultrafuzz-modal-candidate-")); execFileSync("git", ["init", "--quiet"], { cwd: root }); diff --git a/packages/modal/test/smoke.test.ts b/packages/modal/test/smoke.test.ts index c00634c73..5905cb3f9 100644 --- a/packages/modal/test/smoke.test.ts +++ b/packages/modal/test/smoke.test.ts @@ -128,7 +128,6 @@ describe("dedicated cloud command", () => { expect(cliSource.match(/import\("\.\/smoke-modal\.js"\)/gu)).toHaveLength(1); expect(cliSource).not.toMatch(/^import .*smoke-modal/mu); - expect(packageJson.scripts.test).toBe("vitest run"); expect(packageJson.scripts.smoke).toContain("dist/cli.js smoke"); expect(packageJson.scripts.typecheck).toBe( "pnpm --filter @ultrafuzz/modal^... build && tsc -p tsconfig.json --noEmit --pretty false" diff --git a/packages/modal/test/terminal-recovery.integration.test.ts b/packages/modal/test/terminal-recovery.integration.test.ts index 89c3513b8..ded4cdd8c 100644 --- a/packages/modal/test/terminal-recovery.integration.test.ts +++ b/packages/modal/test/terminal-recovery.integration.test.ts @@ -3,7 +3,11 @@ import { tmpdir } from "node:os"; import path from "node:path"; import { createRunLayout, getNodeArtifactDir } from "@ultrafuzz/artifacts"; -import { verifyRequiredArtifactsForAttempt, type PlannedGraphNode } from "@ultrafuzz/runtime"; +import { + verifyRequiredArtifactsForAttempt, + type ArtifactGateAttemptAuthority, + type PlannedGraphNode +} from "@ultrafuzz/runtime"; import { describe, expect, it } from "vitest"; import { @@ -39,7 +43,7 @@ describe("terminal artifact-gate recovery", () => { const notesBefore = fs.readFileSync(notesPath); const findingsBefore = artifactState === "schema-invalid" ? fs.readFileSync(findingsPath) : undefined; - const gate = verifyRequiredArtifactsForAttempt(layout, findingsNode(), "task-one"); + const gate = verifyRequiredArtifactsForAttempt(layout, findingsNode(), "task-one", OUTPUT_ONLY_AUTHORITY); expect(gate.ok).toBe(false); expect(gate.missing).toEqual(artifactState === "missing" ? ["findings.json"] : []); if (artifactState === "schema-invalid") { @@ -220,6 +224,12 @@ function taskWorkflow(attemptId: string, runId = WORKFLOW_RUN_ID) { }; } +/** + * A missing or schema-invalid findings output is rejected from its own bytes, + * before any gate reads the sealed attempt declarations, so none are modeled. + */ +const OUTPUT_ONLY_AUTHORITY = { task: {}, tasks: [] } as unknown as ArtifactGateAttemptAuthority; + /** The strategy node whose only declared output carries the findings contract. */ function findingsNode(): PlannedGraphNode { return { diff --git a/packages/modal/test/worker-lineage.test.ts b/packages/modal/test/worker-lineage.test.ts index f6bc3aae0..54d171cd7 100644 --- a/packages/modal/test/worker-lineage.test.ts +++ b/packages/modal/test/worker-lineage.test.ts @@ -5,7 +5,7 @@ import path from "node:path"; import { afterEach, describe, expect, it } from "vitest"; import { fingerprintModalModel } from "../src/config.js"; -import { DEFAULT_BENCHMARK_MODELS, MODAL_WORKER_LINEAGE_SCHEMA_VERSION } from "../src/defaults.js"; +import { MODAL_WORKER_LINEAGE_SCHEMA_VERSION } from "../src/defaults.js"; import type { ModalWorkerLineage } from "../src/launch-state.js"; import { CheckpointIncompatibleError, @@ -15,6 +15,7 @@ import { readModalWorkerLineage } from "../src/worker-lineage.js"; import { emptyWorkerCheckpoint, runWithTerminalPersistence, WorkerResultWriter } from "../src/worker-result.js"; +import { MODEL_SPEC_FIXTURES } from "./model-spec-fixtures.js"; const roots: string[] = []; @@ -24,7 +25,7 @@ afterEach(() => { describe("persistent Modal worker lineage", () => { it("derives the worker model only from the exact configured lineage fingerprint", () => { - const model = DEFAULT_BENCHMARK_MODELS[0]!; + const model = MODEL_SPEC_FIXTURES[0]; const current = { ...lineage(), model_fingerprint: fingerprintModalModel(model) }; expect(modelForModalWorkerLineage({ models: [model] }, current)).toBe(model); diff --git a/packages/modal/test/worker-result.test.ts b/packages/modal/test/worker-result.test.ts index c27fe4586..d94ebacde 100644 --- a/packages/modal/test/worker-result.test.ts +++ b/packages/modal/test/worker-result.test.ts @@ -12,7 +12,6 @@ import { emptyWorkerCheckpoint, readWorkerCheckpoint, runWithTerminalPersistence, - WORKER_RESULT_ALLOWED_KEYS, WORKER_RESULT_SCHEMA_VERSION, WorkerResultWriter, type WorkerResultContract @@ -232,12 +231,20 @@ describe("strict worker result contracts", () => { }); expect(persistedStatus).toEqual(persistedResult); expect(persistedResult).toEqual(terminal); - expect(Object.keys(persistedResult).every((key) => new Set(WORKER_RESULT_ALLOWED_KEYS).has(key))).toBe( - true - ); - expect(Object.keys(persistedResult).sort()).toEqual( - WORKER_RESULT_ALLOWED_KEYS.filter((key) => key !== "pricing").sort() - ); + expect(Object.keys(persistedResult).sort()).toEqual([ + "attempt", + "checkpoint", + "counts", + "diagnostic_code", + "exit_category", + "generation", + "launch_generation", + "model_work_started", + "result_type", + "runtime_ms", + "schema_version", + "usage" + ]); expect(Object.keys(persistedResult.counts).sort()).toEqual(["failed", "remaining", "succeeded"]); expect(Object.keys(persistedResult.checkpoint).sort()).toEqual(["age_ms", "digest"]); expect(fs.statSync(path.join(root, "result.json")).mode & 0o777).toBe(0o600); diff --git a/packages/modal/test/workspace-config.test.ts b/packages/modal/test/workspace-config.test.ts index 519cdea53..f4080f204 100644 --- a/packages/modal/test/workspace-config.test.ts +++ b/packages/modal/test/workspace-config.test.ts @@ -5,13 +5,13 @@ import { parseProjectConfigToml, resolveConfig } from "@ultrafuzz/config"; import { describe, expect, it } from "vitest"; import { parse } from "yaml"; -import { DEFAULT_BENCHMARK_MODELS } from "../src/defaults.js"; import { PUBLIC_FULL_BENCHMARK_MAX_RUNTIME_SECONDS } from "../src/public-worker.js"; import { modalTargetToml } from "../src/workspace-config.js"; +import { MODEL_SPEC_FIXTURES } from "./model-spec-fixtures.js"; describe("Modal target model profiles", () => { it("overrides both explicit default and benchmark profiles with the selected model", () => { - const model = DEFAULT_BENCHMARK_MODELS[4]!; + const model = MODEL_SPEC_FIXTURES[4]; const config = modalTargetToml(model, 7_200); expect(config).toMatch(/^schema_version = "ultrafuzz\.config\.v2"$/mu); @@ -27,7 +27,7 @@ describe("Modal target model profiles", () => { }); it("selects the packaged smoke audit profile for smoke target preparation", () => { - const config = modalTargetToml(DEFAULT_BENCHMARK_MODELS[0]!, 900, "smoke"); + const config = modalTargetToml(MODEL_SPEC_FIXTURES[0], 900, "smoke"); expect(config).toMatch(/^schema_version = "ultrafuzz\.config\.v2"$/mu); expect(config).toContain('audit_profile = "smoke"'); @@ -39,7 +39,7 @@ describe("Modal target model profiles", () => { }); it("selects the packaged exhaustive audit profile for full-lane target preparation", () => { - const config = modalTargetToml(DEFAULT_BENCHMARK_MODELS[0]!, 1_800, "exhaustive"); + const config = modalTargetToml(MODEL_SPEC_FIXTURES[0], 1_800, "exhaustive"); expect(config).toContain('audit_profile = "exhaustive"'); expect(config).toContain("dynamic_strategies_enumerator = 3"); diff --git a/packages/prompts/package.json b/packages/prompts/package.json index ee13cd892..0b3b04c9b 100644 --- a/packages/prompts/package.json +++ b/packages/prompts/package.json @@ -17,7 +17,7 @@ ], "scripts": { "build": "tsc -p tsconfig.json && rm -rf dist/assets/prompts && mkdir -p dist/assets && cp -R ../../.ultrafuzz/prompts dist/assets/prompts", - "test": "pnpm --filter @ultrafuzz/prompts^... build && vitest run", + "test": "pnpm --filter @ultrafuzz/prompts^... build && vitest run --testTimeout=30000", "typecheck": "pnpm --filter @ultrafuzz/prompts^... build && tsc -p tsconfig.json --noEmit --pretty false" }, "dependencies": { diff --git a/packages/prompts/src/catalog.ts b/packages/prompts/src/catalog.ts index 576a92ca9..d9a39199f 100644 --- a/packages/prompts/src/catalog.ts +++ b/packages/prompts/src/catalog.ts @@ -100,6 +100,20 @@ export function builtInPromptRelativePaths(): string[] { return discoverBuiltInPromptRelativePaths(builtInPromptRoot()); } +/** + * Relative paths of project prompts whose bytes differ from the built-in prompt at the same path. Runs + * use the project copy and `ultrafuzz init` keeps it unless run with `--force`, so after an upgrade such + * a copy is either a deliberate edit or an older release's prompt. + */ +export function projectPromptsDifferingFromBuiltIns(catalog: PromptCatalog): string[] { + const builtInMarkdown = new Map(loadBuiltInPromptAssets().map((asset) => [asset.relativePath, asset.markdown])); + return [...catalog.entries.values()] + .filter((entry) => entry.source === "project" && builtInMarkdown.has(entry.relativePath)) + .filter((entry) => builtInMarkdown.get(entry.relativePath) !== entry.markdown) + .map((entry) => entry.relativePath) + .sort(); +} + function discoverBuiltInPromptRelativePaths(root: string): string[] { return discoverPromptFiles(root) .map((absolutePath) => normalizePromptRelativePath(path.relative(root, absolutePath))) diff --git a/packages/prompts/src/frontmatter.ts b/packages/prompts/src/frontmatter.ts index acc00fbb1..4a2639834 100644 --- a/packages/prompts/src/frontmatter.ts +++ b/packages/prompts/src/frontmatter.ts @@ -1,4 +1,4 @@ -import { parse as parseYaml, stringify as stringifyYaml } from "yaml"; +import { parse as parseYaml } from "yaml"; export const PROMPT_FRONTMATTER_FIELDS = ["id", "display_name"] as const; @@ -12,12 +12,6 @@ export interface PromptFrontmatter { export interface ParsedPromptDocument { frontmatter: PromptFrontmatter; body: string; - rawFrontmatter?: string; - unknownFrontmatter?: Record; -} - -export interface ParsePromptFrontmatterOptions { - allowUnknownFields?: boolean; } export type PromptErrorCode = @@ -27,7 +21,6 @@ export type PromptErrorCode = | "invalid-frontmatter" | "invalid-prompt-path" | "invalid-render-input" - | "invalid-rename" | "missing-template-variable" | "not-ancestor" | "symlink-prompt-path" @@ -47,10 +40,7 @@ export class PromptError extends Error { } } -export function parsePromptFrontmatter( - markdown: string, - options: ParsePromptFrontmatterOptions = {} -): ParsedPromptDocument { +export function parsePromptFrontmatter(markdown: string): ParsedPromptDocument { const frontmatterBlock = readFrontmatterBlock(markdown); if (!frontmatterBlock) { return { @@ -76,16 +66,12 @@ export function parsePromptFrontmatter( throw new PromptError("invalid-frontmatter", "prompt frontmatter must be a YAML mapping"); } - const unknownFrontmatter: Record = {}; for (const key of Object.keys(parsed)) { if (key === "category") { throw new PromptError("invalid-frontmatter", "`category` is no longer supported in prompt frontmatter"); } if (!PROMPT_FRONTMATTER_FIELDS.includes(key as PromptFrontmatterField)) { - if (!options.allowUnknownFields) { - throw new PromptError("invalid-frontmatter", `unsupported prompt frontmatter field: ${key}`); - } - unknownFrontmatter[key] = parsed[key]; + throw new PromptError("invalid-frontmatter", `unsupported prompt frontmatter field: ${key}`); } } @@ -105,59 +91,7 @@ export function parsePromptFrontmatter( } return { frontmatter, - body: frontmatterBlock.body, - rawFrontmatter: frontmatterBlock.raw, - ...(Object.keys(unknownFrontmatter).length > 0 ? { unknownFrontmatter } : {}) - }; -} - -export function serializePromptDocument( - frontmatter: PromptFrontmatter, - body: string, - unknownFrontmatter: Record = {} -): string { - const mapping: Record = {}; - for (const key of PROMPT_FRONTMATTER_FIELDS) { - const value = frontmatter[key]; - if (value !== undefined) { - mapping[key] = value; - } - } - for (const [key, value] of Object.entries(unknownFrontmatter)) { - if (!(key in mapping)) { - mapping[key] = value; - } - } - - if (Object.keys(mapping).length === 0) { - return body; - } - - const yaml = stringifyYaml(mapping, { sortMapEntries: true }).trimEnd(); - const normalizedBody = body.startsWith("\n") ? body.slice(1) : body; - return `---\n${yaml}\n---\n\n${normalizedBody}`; -} - -export interface PromptIdentityDiff { - idChanged: boolean; - displayNameChanged: boolean; - executionIdentityChanged: boolean; - displayNameOnly: boolean; -} - -export function diffPromptIdentity(beforeMarkdown: string, afterMarkdown: string): PromptIdentityDiff { - const before = parsePromptFrontmatter(beforeMarkdown); - const after = parsePromptFrontmatter(afterMarkdown); - - const idChanged = before.frontmatter.id !== after.frontmatter.id; - const displayNameChanged = before.frontmatter.display_name !== after.frontmatter.display_name; - const executionIdentityChanged = idChanged || before.body !== after.body; - - return { - idChanged, - displayNameChanged, - executionIdentityChanged, - displayNameOnly: displayNameChanged && !executionIdentityChanged + body: frontmatterBlock.body }; } diff --git a/packages/prompts/src/index.ts b/packages/prompts/src/index.ts index 50b5497c0..0502495bf 100644 --- a/packages/prompts/src/index.ts +++ b/packages/prompts/src/index.ts @@ -1,5 +1,4 @@ export * from "./catalog.js"; export * from "./frontmatter.js"; export * from "./render.js"; -export * from "./rename.js"; export * from "./scaffold.js"; diff --git a/packages/prompts/src/rename.ts b/packages/prompts/src/rename.ts deleted file mode 100644 index 6f1a20d98..000000000 --- a/packages/prompts/src/rename.ts +++ /dev/null @@ -1,247 +0,0 @@ -import path from "node:path"; -import { parse as parseYaml, stringify as stringifyYaml } from "yaml"; -import { isSafePromptId, parsePromptFrontmatter, PromptError, serializePromptDocument } from "./frontmatter.js"; -import { renamePromptArtifactReferences } from "./render.js"; - -export interface PromptFileSnapshot { - path: string; - contents: string; -} - -export interface PromptTopologyNode { - id: string; - prompt?: string; - dependsOn?: string[]; - depends_on?: string[]; - outputs?: Array<{ path: string; contract: string; primary?: boolean }>; - [key: string]: unknown; -} - -export interface PromptTopologyDocument { - version?: number; - nodes: PromptTopologyNode[]; - [key: string]: unknown; -} - -export interface PromptConcreteNodeSnapshot { - id: string; - logicalId?: string; - logical_id?: string; - dependsOn?: string[]; - depends_on?: string[]; - artifactDir?: string; - artifact_dir?: string; - [key: string]: unknown; -} - -export interface RenamePromptIdInput { - oldId: string; - newId: string; - topology: PromptTopologyDocument | string; - promptFiles: PromptFileSnapshot[]; - concreteNodes?: PromptConcreteNodeSnapshot[]; -} - -export interface RenamePromptIdResult { - topology: PromptTopologyDocument; - topologyYaml?: string; - promptFiles: PromptFileSnapshot[]; - concreteNodes: PromptConcreteNodeSnapshot[]; - concreteIdMap: Record; - renamedPromptPaths: Record; -} - -export function renamePromptId(input: RenamePromptIdInput): RenamePromptIdResult { - validateRename(input.oldId, input.newId); - - const parsedTopology = - typeof input.topology === "string" ? (parseYaml(input.topology) as PromptTopologyDocument) : clone(input.topology); - if (!parsedTopology || !Array.isArray(parsedTopology.nodes)) { - throw new PromptError("invalid-rename", "topology must contain a nodes array"); - } - if (parsedTopology.nodes.some((node) => node.id === input.newId && node.id !== input.oldId)) { - throw new PromptError("invalid-rename", `topology already contains node id \`${input.newId}\``); - } - validatePromptFileRenameConflicts(input.promptFiles, input.oldId, input.newId); - - const renamedPromptPaths: Record = {}; - const promptFiles = input.promptFiles.map((file) => - renamePromptFile(file, input.oldId, input.newId, renamedPromptPaths) - ); - const topology = renameTopology(parsedTopology, input.oldId, input.newId, renamedPromptPaths); - const { concreteNodes, concreteIdMap } = renameConcreteNodes(input.concreteNodes ?? [], input.oldId, input.newId); - - return { - topology, - ...(typeof input.topology === "string" ? { topologyYaml: stringifyYaml(topology, { sortMapEntries: false }) } : {}), - promptFiles, - concreteNodes, - concreteIdMap, - renamedPromptPaths - }; -} - -export function renameConcreteNodeId(concreteId: string, oldId: string, newId: string): string { - if (concreteId === oldId) { - return newId; - } - if (concreteId.startsWith(`${oldId}-`)) { - return `${newId}${concreteId.slice(oldId.length)}`; - } - return concreteId; -} - -export function rewriteArtifactPathForPromptId(value: string, oldId: string, newId: string): string { - return value - .split(/[\\/]/g) - .map((segment) => rewriteArtifactSegment(segment, oldId, newId)) - .join(value.includes("\\") ? "\\" : "/"); -} - -function renamePromptFile( - file: PromptFileSnapshot, - oldId: string, - newId: string, - renamedPromptPaths: Record -): PromptFileSnapshot { - const parsed = parsePromptFrontmatter(file.contents); - const fallbackId = path.basename(file.path).replace(/\.(md|mdx)$/i, ""); - const ownsRenamedId = parsed.frontmatter.id === oldId || fallbackId === oldId; - const nextFrontmatter = { ...parsed.frontmatter }; - if (ownsRenamedId) { - nextFrontmatter.id = newId; - } - - const nextBody = renamePromptArtifactReferences(parsed.body, oldId, newId); - const contents = serializePromptDocument(nextFrontmatter, nextBody, parsed.unknownFrontmatter); - const nextPath = ownsRenamedId ? renamePromptPath(file.path, oldId, newId) : file.path; - if (nextPath !== file.path) { - renamedPromptPaths[file.path] = nextPath; - } - return { - path: nextPath, - contents - }; -} - -function renamePromptPath(promptPath: string, oldId: string, newId: string): string { - const extension = path.extname(promptPath); - const dirname = path.dirname(promptPath); - const basename = path.basename(promptPath, extension); - if (basename !== oldId) { - return promptPath; - } - return path.posix.join(dirname.split(path.sep).join("/"), `${newId}${extension}`); -} - -function renameTopology( - topology: PromptTopologyDocument, - oldId: string, - newId: string, - renamedPromptPaths: Record -): PromptTopologyDocument { - return { - ...topology, - nodes: topology.nodes.map((node) => { - const renamed: PromptTopologyNode = { - ...node, - id: node.id === oldId ? newId : node.id - }; - if (node.prompt) { - renamed.prompt = renamedPromptPaths[node.prompt] ?? renamePromptPath(node.prompt, oldId, newId); - } - if (node.dependsOn) { - renamed.dependsOn = node.dependsOn.map((dependency) => (dependency === oldId ? newId : dependency)); - } - if (node.depends_on) { - renamed.depends_on = node.depends_on.map((dependency) => (dependency === oldId ? newId : dependency)); - } - if (node.outputs) { - renamed.outputs = node.outputs.map((output) => ({ - ...output, - path: rewriteArtifactPathForPromptId(output.path, oldId, newId) - })); - } - return renamed; - }) - }; -} - -function renameConcreteNodes( - concreteNodes: PromptConcreteNodeSnapshot[], - oldId: string, - newId: string -): { concreteNodes: PromptConcreteNodeSnapshot[]; concreteIdMap: Record } { - const concreteIdMap: Record = {}; - const renamed = concreteNodes.map((node) => { - const nextId = renameConcreteNodeId(node.id, oldId, newId); - if (nextId !== node.id) { - concreteIdMap[node.id] = nextId; - } - const next: PromptConcreteNodeSnapshot = { - ...node, - id: nextId - }; - if (node.logicalId === oldId) { - next.logicalId = newId; - } - if (node.logical_id === oldId) { - next.logical_id = newId; - } - if (node.dependsOn) { - next.dependsOn = node.dependsOn.map((dependency) => renameConcreteNodeId(dependency, oldId, newId)); - } - if (node.depends_on) { - next.depends_on = node.depends_on.map((dependency) => renameConcreteNodeId(dependency, oldId, newId)); - } - if (node.artifactDir) { - next.artifactDir = rewriteArtifactPathForPromptId(node.artifactDir, oldId, newId); - } - if (node.artifact_dir) { - next.artifact_dir = rewriteArtifactPathForPromptId(node.artifact_dir, oldId, newId); - } - return next; - }); - - return { - concreteNodes: renamed, - concreteIdMap - }; -} - -function validateRename(oldId: string, newId: string): void { - if (!isSafePromptId(oldId) || !isSafePromptId(newId)) { - throw new PromptError("invalid-rename", "prompt rename IDs must be safe IDs"); - } - if (oldId === newId) { - throw new PromptError("invalid-rename", "prompt rename requires different old and new IDs"); - } -} - -function validatePromptFileRenameConflicts(promptFiles: PromptFileSnapshot[], oldId: string, newId: string): void { - for (const file of promptFiles) { - const parsed = parsePromptFrontmatter(file.contents); - const fallbackId = path.basename(file.path).replace(/\.(md|mdx)$/i, ""); - if (parsed.frontmatter.id === newId || (fallbackId === newId && parsed.frontmatter.id !== oldId)) { - throw new PromptError("invalid-rename", `prompt file already contains id \`${newId}\`: ${file.path}`); - } - } -} - -function rewriteArtifactSegment(segment: string, oldId: string, newId: string): string { - if (segment === oldId) { - return newId; - } - if (segment.startsWith(`${oldId}-`)) { - return `${newId}${segment.slice(oldId.length)}`; - } - const extensionIndex = segment.lastIndexOf("."); - if (extensionIndex > 0 && segment.slice(0, extensionIndex) === oldId) { - return `${newId}${segment.slice(extensionIndex)}`; - } - return segment; -} - -function clone(value: T): T { - return structuredClone(value); -} diff --git a/packages/prompts/src/render.ts b/packages/prompts/src/render.ts index ae91a889a..0a24888b9 100644 --- a/packages/prompts/src/render.ts +++ b/packages/prompts/src/render.ts @@ -1,6 +1,5 @@ -import { mkdirSync, readFileSync, statSync, writeFileSync } from "node:fs"; +import { mkdirSync, readFileSync, writeFileSync } from "node:fs"; import path from "node:path"; -import { fileURLToPath } from "node:url"; import { findingNoteKeyPromptVocabulary, @@ -199,7 +198,6 @@ type ArtifactProducer = export interface TemplateOccurrence { name: string; - rawName: string; start: number; end: number; } @@ -625,36 +623,11 @@ function loadOutputContractTemplate(relativePath: string): string { if (cached !== undefined) { return cached; } - const template = readFileSync(path.join(outputContractTemplateRoot(), relativePath), "utf8"); + const template = readFileSync(path.join(builtInPromptRoot(), "_templates", "output-contract", relativePath), "utf8"); outputContractTemplateCache.set(relativePath, template); return template; } -function outputContractTemplateRoot(): string { - const here = path.dirname(fileURLToPath(import.meta.url)); - const candidates = [ - // This build now copies the prompt tree to dist/assets/prompts, so the - // packaged templates live here. Without this candidate the only path that - // still resolved was the repository fallback below, which exists in a source - // checkout but not in a sealed execution snapshot -- where the package sits - // at modules/@ultrafuzz/prompts/dist and "../../../" reaches modules/. That - // rendered fine at submission and threw the first time a workflow - // re-rendered mid-run, taking the whole run down at WORKFLOW_RENDER_FAILED. - path.join(here, "assets", "prompts", "_templates", "output-contract"), - path.join(here, "prompts", "_templates", "output-contract"), - path.resolve(here, "../../../.ultrafuzz/prompts/_templates/output-contract") - ]; - const found = candidates.find((candidate) => { - try { - return statSync(candidate).isDirectory(); - } catch { - return false; - } - }); - if (found === undefined) throw new Error(`unable to locate the packaged output contract templates from ${here}`); - return found; -} - const agentPreambleTemplateCache = new Map(); /** Load one trusted global-agent prompt fragment from the packaged MDX assets. */ @@ -700,38 +673,6 @@ export function writeRenderedPrompt(result: PromptRenderResult): string { return result.renderedPromptPath; } -export function renamePromptArtifactReferences(template: string, oldLogicalId: string, newLogicalId: string): string { - validateArtifactReferenceId(oldLogicalId); - validateArtifactReferenceId(newLogicalId); - validatePromptVariables(template); - - let rewritten = ""; - let consumed = 0; - for (const occurrence of findTemplateOccurrences(template)) { - rewritten += template.slice(consumed, occurrence.start); - const replacement = renameTemplateVariableName(occurrence.name, oldLogicalId, newLogicalId); - const leadingWhitespace = occurrence.rawName.length - occurrence.rawName.trimStart().length; - const trailingWhitespace = occurrence.rawName.length - occurrence.rawName.trimEnd().length; - rewritten += "{{"; - rewritten += occurrence.rawName.slice(0, leadingWhitespace); - rewritten += replacement; - rewritten += occurrence.rawName.slice(occurrence.rawName.length - trailingWhitespace); - rewritten += "}}"; - - consumed = occurrence.end; - const producer = parseArtifactProducer(occurrence.name); - if (producer?.kind === "current" || (producer?.kind === "logical" && producer.logicalId === oldLogicalId)) { - const suffix = rewriteArtifactPathSuffix(template.slice(consumed), oldLogicalId, newLogicalId); - if (suffix.consumed > 0) { - rewritten += suffix.value; - consumed += suffix.consumed; - } - } - } - rewritten += template.slice(consumed); - return rewritten; -} - export function validateArtifactRelativePath(relativePath: string): void { const parts = relativePath.split("/"); if ( @@ -771,14 +712,12 @@ function findTemplateOccurrences(template: string): TemplateOccurrence[] { if (end === -1) { throw new PromptError("unclosed-template-variable", "template variable is missing a closing delimiter"); } - const rawName = template.slice(start + 2, end); - const name = rawName.trim(); + const name = template.slice(start + 2, end).trim(); if (name === "") { throw new PromptError("empty-template-variable", "template variable name cannot be empty"); } occurrences.push({ name, - rawName, start, end: end + 2 }); @@ -1390,44 +1329,3 @@ function ensureInsidePath(root: string, candidate: string, label: string): void } throw new PromptError("invalid-render-input", `${label} must stay inside ${root}: ${candidate}`); } - -function renameTemplateVariableName(name: string, oldLogicalId: string, newLogicalId: string): string { - if (name === `artifact_path:${oldLogicalId}`) { - return `artifact_path:${newLogicalId}`; - } - if (name === `artifact_handoff:${oldLogicalId}`) { - return `artifact_handoff:${newLogicalId}`; - } - return name; -} - -function rewriteArtifactPathSuffix( - suffix: string, - oldLogicalId: string, - newLogicalId: string -): { value: string; consumed: number } { - if (!suffix.startsWith("/")) { - return { value: "", consumed: 0 }; - } - const end = findSuffixEnd(suffix); - const value = suffix - .slice(0, end) - .split("/") - .map((segment) => rewriteArtifactPathSegment(segment, oldLogicalId, newLogicalId)) - .join("/"); - return { value, consumed: end }; -} - -function rewriteArtifactPathSegment(segment: string, oldLogicalId: string, newLogicalId: string): string { - if (segment === oldLogicalId) { - return newLogicalId; - } - if (segment.startsWith(`${oldLogicalId}-`)) { - return `${newLogicalId}${segment.slice(oldLogicalId.length)}`; - } - const extensionIndex = segment.lastIndexOf("."); - if (extensionIndex > 0 && segment.slice(0, extensionIndex) === oldLogicalId) { - return `${newLogicalId}${segment.slice(extensionIndex)}`; - } - return segment; -} diff --git a/packages/prompts/test/frontmatter.test.ts b/packages/prompts/test/frontmatter.test.ts index 6983ac2a0..7a9aa1ab8 100644 --- a/packages/prompts/test/frontmatter.test.ts +++ b/packages/prompts/test/frontmatter.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "vitest"; -import { diffPromptIdentity, parsePromptFrontmatter, PromptError } from "../src/index.js"; +import { parsePromptFrontmatter, PromptError } from "../src/index.js"; describe("prompt frontmatter", () => { it("parses Markdown-compatible prompt frontmatter", () => { @@ -12,31 +12,11 @@ describe("prompt frontmatter", () => { expect(parsed.body).toBe("# Body"); }); - it("rejects unsupported additional fields by default", () => { + it("rejects unsupported additional fields", () => { expect(() => parsePromptFrontmatter("---\nid: x\nowner: user\n---\nBody")).toThrow(PromptError); }); - it("can preserve documented unknown frontmatter when explicitly allowed", () => { - const parsed = parsePromptFrontmatter("---\nid: x\nowner: user\n---\nBody", { - allowUnknownFields: true - }); - - expect(parsed.unknownFrontmatter).toEqual({ owner: "user" }); - }); - it("rejects removed category frontmatter", () => { expect(() => parsePromptFrontmatter("---\nid: x\ncategory: custom\n---\nBody")).toThrow(/category/); }); - - it("treats display_name changes as labels only", () => { - const before = "---\nid: boundary-tests\ndisplay_name: Boundary Tests\n---\nBody"; - const after = "---\nid: boundary-tests\ndisplay_name: Boundary Tests v2\n---\nBody"; - - expect(diffPromptIdentity(before, after)).toEqual({ - idChanged: false, - displayNameChanged: true, - executionIdentityChanged: false, - displayNameOnly: true - }); - }); }); diff --git a/packages/prompts/test/rename.test.ts b/packages/prompts/test/rename.test.ts deleted file mode 100644 index 1030a79c6..000000000 --- a/packages/prompts/test/rename.test.ts +++ /dev/null @@ -1,95 +0,0 @@ -import { describe, expect, it } from "vitest"; -import { diffPromptIdentity, renamePromptId } from "../src/index.js"; - -describe("prompt ID rename sync", () => { - it("synchronizes topology ids, dependencies, prompt files, artifact refs, and concrete ids", () => { - const result = renamePromptId({ - oldId: "boundary-tests", - newId: "edge-tests", - topology: { - version: 1, - nodes: [ - { - id: "boundary-tests", - prompt: "strategies/boundary-tests.mdx", - depends_on: ["base-test-setup"], - outputs: [ - { - path: "boundary-tests/generated-tests.json", - contract: "ultrafuzz/generated-tests@3", - primary: true - } - ] - }, - { - id: "dedupe-findings", - prompt: "review/dedupe-findings.mdx", - depends_on: ["boundary-tests"] - } - ] - }, - promptFiles: [ - { - path: "strategies/boundary-tests.mdx", - contents: - "---\nid: boundary-tests\ndisplay_name: Boundary Tests\n---\nRead {{artifact_handoff:base-test-setup}} and {{artifact_path:boundary-tests}}/boundary-tests.json" - }, - { - path: "review/dedupe-findings.mdx", - contents: - "---\nid: dedupe-findings\n---\nUse {{artifact_handoff:boundary-tests}} and {{artifact_path:boundary-tests}}/generated-tests.json" - } - ], - concreteNodes: [ - { - id: "boundary-tests-0", - logicalId: "boundary-tests", - artifactDir: "/runs/run-1/artifacts/boundary-tests-0" - }, - { - id: "dedupe-findings", - logicalId: "dedupe-findings", - dependsOn: ["boundary-tests-0"] - } - ] - }); - - expect(result.topology.nodes[0]?.id).toBe("edge-tests"); - expect(result.topology.nodes[0]?.prompt).toBe("strategies/edge-tests.mdx"); - expect(result.topology.nodes[0]?.outputs?.map((output) => output.path)).toEqual([ - "edge-tests/generated-tests.json" - ]); - expect(result.topology.nodes[1]?.depends_on).toEqual(["edge-tests"]); - expect(result.promptFiles[0]?.path).toBe("strategies/edge-tests.mdx"); - expect(result.promptFiles[0]?.contents).toContain("id: edge-tests"); - expect(result.promptFiles[0]?.contents).toContain("{{artifact_path:edge-tests}}/edge-tests.json"); - expect(result.promptFiles[1]?.contents).toContain("{{artifact_handoff:edge-tests}}"); - expect(result.concreteIdMap).toEqual({ "boundary-tests-0": "edge-tests-0" }); - expect(result.concreteNodes[0]?.artifactDir).toBe("/runs/run-1/artifacts/edge-tests-0"); - expect(result.concreteNodes[1]?.dependsOn).toEqual(["edge-tests-0"]); - }); - - it("rejects frontmatter rename conflicts", () => { - expect(() => - renamePromptId({ - oldId: "boundary-tests", - newId: "edge-tests", - topology: { - version: 1, - nodes: [{ id: "boundary-tests", prompt: "strategies/boundary-tests.md" }] - }, - promptFiles: [ - { path: "strategies/boundary-tests.md", contents: "---\nid: boundary-tests\n---\nBody" }, - { path: "strategies/edge-tests.md", contents: "---\nid: edge-tests\n---\nBody" } - ] - }) - ).toThrow(/already contains id/); - }); - - it("does not change execution identity for display_name-only edits", () => { - const before = "---\nid: boundary-tests\ndisplay_name: Boundary Tests\n---\nBody"; - const after = "---\nid: boundary-tests\ndisplay_name: Edge Boundary Tests\n---\nBody"; - - expect(diffPromptIdentity(before, after).displayNameOnly).toBe(true); - }); -}); diff --git a/packages/prompts/test/render.test.ts b/packages/prompts/test/render.test.ts index 8921c2ae9..6f59c11c7 100644 --- a/packages/prompts/test/render.test.ts +++ b/packages/prompts/test/render.test.ts @@ -1,15 +1,4 @@ -import { - copyFileSync, - cpSync, - existsSync, - mkdirSync, - mkdtempSync, - readFileSync, - realpathSync, - rmSync, - symlinkSync -} from "node:fs"; -import { execFileSync } from "node:child_process"; +import { cpSync, existsSync, mkdirSync, mkdtempSync, readFileSync, realpathSync, rmSync, symlinkSync } from "node:fs"; import os from "node:os"; import path from "node:path"; import { fileURLToPath, pathToFileURL } from "node:url"; @@ -208,42 +197,6 @@ describe("prompt rendering", () => { const rendered = packaged.renderPrompt(input).renderedMarkdown; expect(rendered).toContain("For every output declared with `Contract: ultrafuzz/findings@2`"); - }); - - it("renders output-contract guidance and prompt partials from an installed package layout", async () => { - const tmp = mkdtempSync(path.join(realpathSync(os.tmpdir()), "ufz-installed-prompts-")); - tmpDirs.push(tmp); - const repoRoot = fileURLToPath(new URL("../../..", import.meta.url)); - const appRoot = path.join(tmp, "app"); - const packageRoot = path.join(appRoot, "node_modules", "@ultrafuzz", "prompts"); - const distRoot = path.join(packageRoot, "dist"); - mkdirSync(packageRoot, { recursive: true }); - execFileSync( - "pnpm", - [ - "exec", - "tsc", - "-p", - path.join(repoRoot, "packages", "prompts", "tsconfig.json"), - "--outDir", - distRoot, - "--tsBuildInfoFile", - path.join(distRoot, ".tsbuildinfo") - ], - { cwd: repoRoot, stdio: "pipe" } - ); - cpSync(path.join(repoRoot, ".ultrafuzz", "prompts"), path.join(distRoot, "prompts"), { recursive: true }); - copyFileSync(path.join(repoRoot, "packages", "prompts", "package.json"), path.join(packageRoot, "package.json")); - linkPromptDependencies(repoRoot, path.join(appRoot, "node_modules")); - - const installed = (await import( - `${pathToFileURL(path.join(distRoot, "render.js")).href}?installed-layout=${Date.now()}` - )) as { renderPrompt: typeof renderPrompt }; - const input = baseRenderInput(tmp); - input.prompt = `${input.prompt}\n{{coverage_evidence_markdown_projection}}`; - const rendered = installed.renderPrompt(input).renderedMarkdown; - - expect(rendered).toContain("For every output declared with `Contract: ultrafuzz/findings@2`"); expect(rendered).toContain("Validation command: `ultrafuzz json validate --schema"); expect(rendered).toContain("Contract validation command: `ultrafuzz artifact validate"); expect(rendered).toContain("- : `/`"); diff --git a/packages/runtime/package.json b/packages/runtime/package.json index e80034ba5..29285e072 100644 --- a/packages/runtime/package.json +++ b/packages/runtime/package.json @@ -18,7 +18,6 @@ "scripts": { "build": "tsc -p tsconfig.json && cp ../../.ultrafuzz/topology.yml dist/topology.yml && rm -rf dist/templates && cp -R src/templates dist/templates && rm -rf dist/schema && cp -R schema dist/schema && node scripts/verify-schema-registry.mjs", "test": "pnpm --filter @ultrafuzz/runtime... build && rm -rf dist-test && tsc -p tsconfig.test.json && node scripts/run-tests.mjs && bun test dist-test/test/runtime.test.js --test-name-pattern '^Bun adapter contract:'", - "test:pr-smoke:prebuilt": "rm -rf dist-test && tsc -p tsconfig.test.json && node scripts/run-pr-smoke-tests.mjs", "test:release:runtime-shard": "pnpm --filter @ultrafuzz/runtime... build && rm -rf dist-test && tsc -p tsconfig.test.json && node scripts/run-runtime-test-shard.mjs", "test:release:supporting": "pnpm --filter @ultrafuzz/runtime... build && rm -rf dist-test && tsc -p tsconfig.test.json && node scripts/run-tests.mjs supporting && bun test dist-test/test/runtime.test.js --test-name-pattern '^Bun adapter contract:'", "test:kimi-contract": "pnpm --filter @ultrafuzz/runtime... build && rm -rf dist-test && tsc -p tsconfig.test.json && bun test dist-test/test/runtime.test.js --test-name-pattern '^Bun adapter contract: generated Kimi'", @@ -37,6 +36,7 @@ "mdast-util-from-markdown": "2.0.3", "npm": "11.19.0", "proper-lockfile": "4.1.2", + "smol-toml": "1.7.1", "typescript": "6.0.3", "zod": "4.4.3" }, diff --git a/packages/runtime/scripts/run-pr-smoke-tests.mjs b/packages/runtime/scripts/run-pr-smoke-tests.mjs deleted file mode 100644 index 5b6c249df..000000000 --- a/packages/runtime/scripts/run-pr-smoke-tests.mjs +++ /dev/null @@ -1,159 +0,0 @@ -import { spawnSync } from "node:child_process"; -import { readFileSync } from "node:fs"; - -const supportingTestFiles = [ - "dist-test/test/agent-adapter-boundaries.test.js", - "dist-test/test/runtime-test-shard.test.js", - "dist-test/test/report-publication-status.test.js", - "dist-test/test/source-revision.test.js", - "dist-test/test/workflow-control.test.js" -]; - -const namedTests = new Map([ - [ - "test/runtime.test.ts", - [ - "init resolves one-hour node and execution-resource timeout defaults", - "validate rejects unknown agent references before launch", - "plan creates run layout, graph fingerprint, and rendered prompt before Smithers submission", - "compileSmithersWorkflow gates native dependencies on deterministic artifact verification", - "compileSmithersWorkflow maps cloud attempts to portable provider sandboxes", - "resume, replay, and fork delegate linked runs to Smithers lifecycle verbs", - "ordinary resume checks active-run ownership before detached preflight", - "a refresh resume reuses its own ownership inspection instead of inspecting twice", - "native continuation does not use historical trusted CLI identity as an authorization gate", - // Reads the release that is actually pinned. This is the lane's tripwire - // for a runner bump that widens an enum or the inspect envelope, which - // otherwise only shows up as a failed production run. - "pinned runner state and envelope contracts match Ultrafuzz's mirrors", - "the pinned runner drops resume pointers only from dead attempts" - ] - ], - [ - "test/generated-workflow-verifier.test.ts", - ["generated workflow input is an exact current-only envelope with bounded JSON operator data"] - ] -]); -const bunTestFile = "test/runtime.test.ts"; -const bunTestNamePrefix = "Bun adapter contract: "; -const bunTestNames = [ - "generated DeepSeek adapter uses the official endpoint and preserves independent usage components", - "generated DeepSeek adapter cleans an upstream command when environment policy rejects it", - "generated DeepSeek adapter corrects Smithers result and failed-attempt telemetry", - "generated DeepSeek adapter rejects ambiguous or noncanonical result telemetry", - // Needs bun:sqlite, so it can only run in this lane. Gates the claim that the - // pinned runner's schema migrations are additive over a stopped 0.34.0 store. - "pinned store migrations are additive over a 0.34.0 database" -]; -const selectedBunTestNames = bunTestNames.map((name) => `${bunTestNamePrefix}${name}`); -const smokeEnvironment = { ...process.env }; -delete smokeEnvironment.ULTRAFUZZ_RUNTIME_TEST_SHARD; -const selectedTestNames = [...namedTests.values()].flat(); - -for (const [sourcePath, names] of namedTests) { - const source = readFileSync(sourcePath, "utf8"); - for (const name of names) { - if (!source.includes(`test(${JSON.stringify(name)}`)) { - throw new Error(`PR runtime smoke test is not registered: ${name}`); - } - } -} -const bunTestSource = readFileSync(bunTestFile, "utf8"); -for (const name of bunTestNames) { - if (!bunTestSource.includes(JSON.stringify(name))) { - throw new Error(`PR Bun runtime smoke test is not registered: ${name}`); - } -} - -runNodeTests(supportingTestFiles); -runNodeTests( - [...namedTests.keys()].map((sourcePath) => `dist-test/${sourcePath.replace(/\.ts$/u, ".js")}`), - selectedTestNames, - selectedTestNames -); -runBunTests(`dist-test/${bunTestFile.replace(/\.ts$/u, ".js")}`, selectedBunTestNames); - -function runNodeTests(files, testNames, expectedTestNames) { - const args = ["--test", "--test-reporter=tap"]; - if (testNames !== undefined) { - const pattern = `^(?:${testNames.map(escapeRegExp).join("|")})$`; - args.push(`--test-name-pattern=${pattern}`); - } - args.push(...files); - - const result = spawnSync(process.execPath, args, { - encoding: "utf8", - env: smokeEnvironment, - stdio: ["inherit", "pipe", "pipe"] - }); - process.stdout.write(result.stdout ?? ""); - process.stderr.write(result.stderr ?? ""); - if (result.error !== undefined) throw result.error; - if (result.status !== 0) process.exit(result.status ?? 1); - if (expectedTestNames !== undefined) assertNamedTestsPassed(result.stdout ?? "", expectedTestNames); -} - -function runBunTests(file, testNames) { - const pattern = `^(?:${testNames.map(escapeRegExp).join("|")})$`; - const result = spawnSync("bun", ["test", file, "--test-name-pattern", pattern], { - encoding: "utf8", - env: smokeEnvironment, - stdio: ["inherit", "pipe", "pipe"] - }); - process.stdout.write(result.stdout ?? ""); - process.stderr.write(result.stderr ?? ""); - if (result.error !== undefined) throw result.error; - if (result.status !== 0) process.exit(result.status ?? 1); - assertBunTestsPassed(`${result.stdout ?? ""}\n${result.stderr ?? ""}`, testNames); -} - -function escapeRegExp(value) { - return value.replace(/[.*+?^${}()|[\]\\]/gu, "\\$&"); -} - -function assertNamedTestsPassed(output, expectedTestNames) { - const passedTestNames = [...output.matchAll(/^ok [0-9]+ - (.+)$/gmu)] - .map((match) => match[1]) - .filter((name) => !name.includes(" # SKIP") && !name.includes(" # TODO")) - .sort(); - const expected = [...expectedTestNames].sort(); - if (passedTestNames.length !== expected.length || passedTestNames.some((name, index) => name !== expected[index])) { - throw new Error(`PR runtime smoke passed unexpected tests: ${JSON.stringify(passedTestNames)}`); - } -} - -// Bun's runner prints a `(fail) ` line per failure but no per-test line -// for a pass -- verified against the pinned `bun-version: 1.3.14` in ci.yml, and -// against 1.4.0. So the post-condition is read off the run summary instead: the -// name pattern selects exactly `expectedTestNames`, so a rename, a skip or a -// filtered-out test shows up as a pass count below the expected one, which is -// the property this check exists to enforce. `stripAnsi` because bun colourises -// the summary whenever it believes it has a terminal. -function assertBunTestsPassed(output, expectedTestNames) { - const plain = stripAnsi(output); - const failures = [...plain.matchAll(/^\(fail\) (.+?)(?: \[[^\]]*\])?$/gmu)].map((match) => match[1]); - if (failures.length > 0) { - throw new Error(`PR Bun runtime smoke failed tests: ${JSON.stringify(failures)}`); - } - const summary = (label) => { - const matches = [...plain.matchAll(new RegExp(`^\\s*(\\d+) ${label}$`, "gmu"))].map((match) => Number(match[1])); - if (matches.length !== 1) { - throw new Error(`PR Bun runtime smoke could not read the "${label}" count from the runner summary`); - } - return matches[0]; - }; - const passed = summary("pass"); - const failed = summary("fail"); - if (failed !== 0) throw new Error(`PR Bun runtime smoke reported ${failed} failing test(s)`); - if (passed !== expectedTestNames.length) { - throw new Error( - `PR Bun runtime smoke ran ${passed} of ${expectedTestNames.length} expected tests; one was renamed, skipped, or filtered out` - ); - } -} - -function stripAnsi(value) { - // Built with `new RegExp` rather than a literal: the CSI introducer is a - // control character, which a regex literal cannot carry past `no-control-regex`. - return value.replace(new RegExp(`${String.fromCharCode(27)}\\[[0-9;]*m`, "gu"), ""); -} diff --git a/packages/runtime/scripts/run-tests.mjs b/packages/runtime/scripts/run-tests.mjs index 4cf057a0b..8eb8d5b61 100644 --- a/packages/runtime/scripts/run-tests.mjs +++ b/packages/runtime/scripts/run-tests.mjs @@ -12,13 +12,7 @@ const selectorFiles = new Map([ .filter((entry) => entry.endsWith(".test.js") && entry !== "runtime.test.js") .sort() .map((entry) => path.join("dist-test/test", entry)) - ], - [ - "materialize", - ["dist-test/test/materialize.test.js", "dist-test/test/clean.test.js", "dist-test/test/runtime.test.js"] - ], - ["clean", ["dist-test/test/clean.test.js", "dist-test/test/runtime.test.js"]], - ["smithers", ["dist-test/test/runtime.test.js"]] + ] ]); const testFiles = diff --git a/packages/runtime/src/agent-registry.ts b/packages/runtime/src/agent-registry.ts index d1d655818..2d37da331 100644 --- a/packages/runtime/src/agent-registry.ts +++ b/packages/runtime/src/agent-registry.ts @@ -1,9 +1,15 @@ +import { createRequire } from "node:module"; import path from "node:path"; import { readSinglyLinkedRegularFileSnapshotInside } from "@ultrafuzz/artifacts"; -import * as ts from "typescript"; +import type * as TypeScript from "typescript"; import { errorMessage } from "@ultrafuzz/artifacts"; +// Required by analyzeAgentRegistry rather than imported: every ultrafuzz CLI process imports this +// package's index, including each agent's `json validate` call, and only `validate` and `init` analyze a +// registry. (Smithers engine processes load TypeScript through smthrs anyway.) +let ts: typeof TypeScript; + export const AGENT_REGISTRY_RELATIVE_PATH = ".smithers/agents/index.ts"; const MAX_AGENT_REGISTRY_BYTES = 256 * 1024; const MAX_ANALYSIS_DEPTH = 64; @@ -50,6 +56,7 @@ export function agentRegistryRegisters(inspection: AgentRegistryInspection, agen const SAFE_AGENT_REF_PATTERN = /^(?!.*\.\.)[A-Za-z_][A-Za-z0-9_.:-]{0,127}$/u; function analyzeAgentRegistry(sourceText: string): ReadonlySet { + ts = createRequire(import.meta.url)("typescript") as typeof TypeScript; const source = ts.createSourceFile( AGENT_REGISTRY_RELATIVE_PATH, sourceText, @@ -57,7 +64,8 @@ function analyzeAgentRegistry(sourceText: string): ReadonlySet { false, ts.ScriptKind.TS ); - const parseDiagnostics = (source as ts.SourceFile & { parseDiagnostics: readonly ts.Diagnostic[] }).parseDiagnostics; + const parseDiagnostics = (source as TypeScript.SourceFile & { parseDiagnostics: readonly TypeScript.Diagnostic[] }) + .parseDiagnostics; if (parseDiagnostics.length > 0) throw new Error("agent registry contains invalid TypeScript syntax"); const exported = exportedAgentFactoriesLocalNames(source); if (exported.size !== 1) { @@ -65,10 +73,10 @@ function analyzeAgentRegistry(sourceText: string): ReadonlySet { } const bindings = topLevelConstBindings(source); const objectIsShadowed = topLevelNameIsBound(source, "Object"); - const memo = new Map | null>(); + const memo = new Map | null>(); let steps = 0; - const evaluate = (expression: ts.Expression, depth: number): ReadonlyMap | undefined => { + const evaluate = (expression: TypeScript.Expression, depth: number): ReadonlyMap | undefined => { steps += 1; if (steps > MAX_ANALYSIS_STEPS) throw new Error("agent registry static analysis exceeded its step budget"); if (depth > MAX_ANALYSIS_DEPTH) throw new Error("agent registry static analysis exceeded its depth budget"); @@ -127,14 +135,14 @@ function analyzeAgentRegistry(sourceText: string): ReadonlySet { } function objectMemberIsUsableFactory( - property: ts.ObjectLiteralElementLike, - bindings: ReadonlyMap, + property: TypeScript.ObjectLiteralElementLike, + bindings: ReadonlyMap, visiting: ReadonlySet, depth: number ): boolean { if (depth > MAX_ANALYSIS_DEPTH) throw new Error("agent registry factory analysis exceeded its depth budget"); if (ts.isMethodDeclaration(property)) return true; - let expression: ts.Expression | undefined; + let expression: TypeScript.Expression | undefined; if (ts.isPropertyAssignment(property)) expression = property.initializer; else if (ts.isShorthandPropertyAssignment(property)) expression = property.objectAssignmentInitializer ?? property.name; @@ -149,8 +157,8 @@ function objectMemberIsUsableFactory( } function factoryExpressionIsNonNullish( - expression: ts.Expression, - bindings: ReadonlyMap, + expression: TypeScript.Expression, + bindings: ReadonlyMap, visiting: ReadonlySet, depth: number ): boolean { @@ -164,7 +172,7 @@ function factoryExpressionIsNonNullish( return factoryExpressionIsNonNullish(initializer, bindings, new Set([...visiting, current.text]), depth + 1); } -function exportedAgentFactoriesLocalNames(source: ts.SourceFile): ReadonlySet { +function exportedAgentFactoriesLocalNames(source: TypeScript.SourceFile): ReadonlySet { const localNames = new Set(); for (const statement of source.statements) { if (ts.isVariableStatement(statement) && hasExportModifier(statement)) { @@ -191,8 +199,8 @@ function exportedAgentFactoriesLocalNames(source: ts.SourceFile): ReadonlySet { - const bindings = new Map(); +function topLevelConstBindings(source: TypeScript.SourceFile): ReadonlyMap { + const bindings = new Map(); for (const statement of source.statements) { if (!ts.isVariableStatement(statement) || (statement.declarationList.flags & ts.NodeFlags.Const) === 0) continue; for (const declaration of statement.declarationList.declarations) { @@ -203,25 +211,10 @@ function topLevelConstBindings(source: ts.SourceFile): ReadonlyMap entry.name.text === name) - ) - return true; - } + if (ts.isImportDeclaration(statement) && importClauseBindsName(statement.importClause, name)) return true; if ( ts.isVariableStatement(statement) && statement.declarationList.declarations.some((declaration) => bindingNameContains(declaration.name, name)) @@ -235,7 +228,14 @@ function topLevelNameIsBound(source: ts.SourceFile, name: string): boolean { return false; } -function bindingNameContains(binding: ts.BindingName, name: string): boolean { +function importClauseBindsName(clause: TypeScript.ImportClause | undefined, name: string): boolean { + if (clause?.name?.text === name) return true; + const bindings = clause?.namedBindings; + if (bindings !== undefined && ts.isNamespaceImport(bindings)) return bindings.name.text === name; + return bindings !== undefined && bindings.elements.some((entry) => entry.name.text === name); +} + +function bindingNameContains(binding: TypeScript.BindingName, name: string): boolean { if (ts.isIdentifier(binding)) return binding.text === name; return binding.elements.some( (element) => !ts.isOmittedExpression(element) && bindingNameContains(element.name, name) @@ -243,9 +243,9 @@ function bindingNameContains(binding: ts.BindingName, name: string): boolean { } function isUnshadowedObjectFreeze( - expression: ts.Expression, + expression: TypeScript.Expression, objectIsShadowed: boolean -): expression is ts.CallExpression { +): expression is TypeScript.CallExpression { return ( !objectIsShadowed && ts.isCallExpression(expression) && @@ -257,18 +257,18 @@ function isUnshadowedObjectFreeze( ); } -function hasExportModifier(node: ts.VariableStatement): boolean { +function hasExportModifier(node: TypeScript.VariableStatement): boolean { return ts.getModifiers(node)?.some((modifier) => modifier.kind === ts.SyntaxKind.ExportKeyword) ?? false; } -function unwrapTypeExpressions(expression: ts.Expression): ts.Expression { +function unwrapTypeExpressions(expression: TypeScript.Expression): TypeScript.Expression { let current = expression; while (ts.isAsExpression(current) || ts.isSatisfiesExpression(current) || ts.isParenthesizedExpression(current)) current = current.expression; return current; } -function objectMemberName(property: ts.ObjectLiteralElementLike): string | undefined { +function objectMemberName(property: TypeScript.ObjectLiteralElementLike): string | undefined { if ( !ts.isPropertyAssignment(property) && !ts.isShorthandPropertyAssignment(property) && diff --git a/packages/runtime/src/aggregation-semantic-context.ts b/packages/runtime/src/aggregation-semantic-context.ts index abeca8c00..c62c72a4a 100644 --- a/packages/runtime/src/aggregation-semantic-context.ts +++ b/packages/runtime/src/aggregation-semantic-context.ts @@ -1,137 +1,85 @@ -import crypto from "node:crypto"; -import path from "node:path"; -import { isDeepStrictEqual } from "node:util"; - import { - ARTIFACT_MANIFEST_FILE, - ARTIFACT_VERIFICATION_SCHEMA_VERSION, - MAX_GENERATED_TEST_COMPANION_BYTES, - assertArtifactVerificationMarkerSemantics, - assertGeneratedTestManifestSemantics, - assertNoSymlinkComponents, - assertPathInside, - parseStrictJsonBytes, - readPlannedGraphDocument, - readSinglyLinkedRegularFileSnapshotInside, safeResolveInside, - validateArtifactContractBytes, - validateArtifactManifest, - validateArtifactVerificationMarker, - type ArtifactManifest, - type ArtifactVerificationMarker, type GeneratedTestEntry, type GeneratedTestManifest, - type PlannedGraphNodeDocument, type RunLayout, type SemanticAggregationContext, type SemanticAggregationSourceBundleContext, type SemanticAggregationSourceEntryContext } from "@ultrafuzz/artifacts"; -const MAX_AGGREGATION_AUTHORITY_BYTES = 64 * 1024 * 1024; -const MAX_AGGREGATION_PREREQUISITE_MANIFESTS = 4_096; -const MAX_AGGREGATION_PREREQUISITE_EDGES = 16_384; -const MAX_AGGREGATION_PREREQUISITE_BYTES = 64 * 1024 * 1024; -const VERIFICATION_MARKER_DIRECTORY = ".ultrafuzz-verification"; - -interface AuthenticatedPublicationSnapshot { - path: string; - absolutePath: string; - bytes: Buffer; - sha256: string; -} - -interface AuthenticatedProducerAuthority { - marker: ArtifactVerificationMarker; - artifactManifest: ArtifactManifest; - artifactManifestBytes: Buffer; - publications: ReadonlyMap; -} +import type { PlannedGraphNode } from "./types.js"; +import type { VerifiedNodeOutputSnapshot, VerifiedOutputArtifactSnapshot } from "./verified-output.js"; -interface PlannedAttemptAuthority { - node: PlannedGraphNodeDocument; +/** A finalized `ultrafuzz/generated-tests@3` producer the aggregating attempt admitted. */ +export interface AggregationSourceProducer { attemptId: string; - attemptIndex: number; + node: Pick; + authority: Pick; + outputs: readonly VerifiedOutputArtifactSnapshot[]; } /** - * Build the aggregation gate's immutable source authority from the planned - * graph's exact transitive ancestors and their current verifier publications. + * Build the aggregation gate's source authority from the producers the caller + * resolved through the aggregating attempt's sealed ancestor closure and + * verifier-persisted dependency admission. Those are the producers the + * in-workflow verifier admitted, so an optional producer that failed is absent + * rather than fatal, and each source is keyed by its sealed attempt ID. */ export function authenticatedAggregationSemanticContext(input: { layout: RunLayout; - node: PlannedGraphNodeDocument; attemptId: string; + producers: readonly AggregationSourceProducer[]; }): SemanticAggregationContext { - const graph = readPlannedGraphDocument(input.layout.graphPath); - const current = graph.nodes.find((candidate) => candidate.id === input.node.id); - if (current === undefined || !isDeepStrictEqual(current, input.node)) { - throw new Error(`aggregation node ${JSON.stringify(input.node.id)} does not match the current planned graph`); - } - const sourceBundles: SemanticAggregationSourceBundleContext[] = []; - for (const producer of transitiveAncestorNodes(current, graph.nodes)) { - if (producer.kind !== "agentic") continue; - const generatedOutputs = producer.outputs.filter((output) => output.contract === "ultrafuzz/generated-tests@3"); - if (generatedOutputs.length === 0) continue; - for (const attempt of plannedAttempts(producer)) { - const artifactRoot = safeResolveInside( - input.layout.artifactsDir, - attempt.attemptId, - "generated-test aggregation source artifact directory" - ); - const authority = authenticatedProducerAuthority( - input.layout, - graph.nodes, - producer, - attempt.attemptId, - attempt.attemptIndex, - artifactRoot - ); - for (const output of generatedOutputs) { - const markerArtifact = authority.marker.artifacts.find((artifact) => artifact.path === output.path); - if (markerArtifact === undefined) { - throw new Error(`verified producer ${attempt.attemptId} omits generated-test output ${output.path}`); - } - const manifestSnapshot = authority.publications.get(output.path); - if (manifestSnapshot === undefined || markerArtifact.sha256 !== manifestSnapshot.sha256) { - throw new Error(`verified generated-test manifest publication changed ${attempt.attemptId}/${output.path}`); - } - const validation = validateArtifactContractBytes( - "ultrafuzz/generated-tests@3", - manifestSnapshot.bytes, - manifestSnapshot.absolutePath - ); - if (!validation.ok || validation.value === undefined) { - throw new Error(`verified generated-test manifest is no longer valid ${attempt.attemptId}/${output.path}`); - } - const manifest = validation.value as GeneratedTestManifest; - assertGeneratedTestManifestSemantics(manifest); - if (manifest.run_id !== input.layout.runId || manifest.node_id !== producer.logical_id) { - throw new Error(`verified generated-test manifest identity changed ${attempt.attemptId}/${output.path}`); - } - const entries = [ - ...manifest.generated_tests.map((entry) => - authenticatedEntry("generated-test", entry, authority.publications) - ), - ...manifest.support_files.map((entry) => authenticatedEntry("support-file", entry, authority.publications)) - ]; - sourceBundles.push( - Object.freeze({ - strategy: producer.logical_id, - nodeId: manifest.node_id, - sourceAttemptId: attempt.attemptId, - attemptIndex: attempt.attemptIndex, - sourceManifestPath: manifestSnapshot.absolutePath, - sourceManifestRelativePath: output.path, - sourceManifestSha256: manifestSnapshot.sha256, - sourceRunId: manifest.run_id, - framework: manifest.framework, - entries: Object.freeze(entries) - }) + const sourceBundles = input.producers.flatMap((producer) => { + const publications = new Map(producer.authority.publications.map((publication) => [publication.path, publication])); + const entry = ( + kind: SemanticAggregationSourceEntryContext["kind"], + candidate: GeneratedTestEntry + ): SemanticAggregationSourceEntryContext => { + const publication = publications.get(candidate.path); + if ( + publication === undefined || + publication.sha256 !== candidate.sha256 || + publication.bytes.byteLength !== candidate.size_bytes + ) { + throw new Error( + `verified generated-test companion publication changed ${producer.attemptId}/${candidate.path}` ); } - } - } + return Object.freeze({ + kind, + sourceArtifactPath: publication.absolute_path, + sourceRelativePath: candidate.path, + sizeBytes: candidate.size_bytes, + sha256: candidate.sha256, + bytes: Buffer.from(publication.bytes), + ...(candidate.language === undefined ? {} : { language: candidate.language }), + ...(candidate.description === undefined ? {} : { description: candidate.description }), + ...(candidate.provenance === undefined ? {} : { provenance: Object.freeze({ ...candidate.provenance }) }) + }); + }; + return producer.outputs.map((output): SemanticAggregationSourceBundleContext => { + const manifest = output.value as GeneratedTestManifest; + return Object.freeze({ + strategy: producer.node.logical_id, + nodeId: manifest.node_id, + sourceAttemptId: producer.attemptId, + // The verifier attributes a bundle to its producer's loop attempt index, + // which the sealed task manifest binds to the planned node's loop. + attemptIndex: producer.node.loop.attempt_index, + sourceManifestPath: output.absolute_path, + sourceManifestRelativePath: output.path, + sourceManifestSha256: output.sha256, + sourceRunId: manifest.run_id, + framework: manifest.framework, + entries: Object.freeze([ + ...manifest.generated_tests.map((candidate) => entry("generated-test", candidate)), + ...manifest.support_files.map((candidate) => entry("support-file", candidate)) + ]) + }); + }); + }); sourceBundles.sort( (left, right) => left.sourceAttemptId.localeCompare(right.sourceAttemptId) || @@ -142,342 +90,3 @@ export function authenticatedAggregationSemanticContext(input: { sourceBundles: Object.freeze(sourceBundles) }); } - -function transitiveAncestorNodes( - current: PlannedGraphNodeDocument, - nodes: readonly PlannedGraphNodeDocument[] -): PlannedGraphNodeDocument[] { - const byId = new Map(nodes.map((node) => [node.id, node])); - const ancestors = new Map(); - const pending = [...current.depends_on]; - while (pending.length > 0) { - const id = pending.pop()!; - if (ancestors.has(id)) continue; - const node = byId.get(id); - if (node === undefined) throw new Error(`aggregation dependency is absent from the planned graph: ${id}`); - ancestors.set(id, node); - pending.push(...node.depends_on); - } - return [...ancestors.values()].sort((left, right) => left.id.localeCompare(right.id)); -} - -function plannedAttempts(node: PlannedGraphNodeDocument): Array<{ attemptId: string; attemptIndex: number }> { - if (node.model_fanout.length <= 1) { - return [{ attemptId: node.id, attemptIndex: node.model_fanout[0]?.attempt_index ?? node.loop.attempt_index }]; - } - return node.model_fanout.map((model) => ({ - attemptId: `${node.id}__model_${model.model_index}__attempt_${model.attempt_index}`, - attemptIndex: model.attempt_index - })); -} - -function plannedAttemptAuthorities(nodes: readonly PlannedGraphNodeDocument[]): Map { - const authorities = new Map(); - for (const node of nodes) { - for (const attempt of plannedAttempts(node)) { - if (authorities.has(attempt.attemptId)) { - throw new Error(`planned graph repeats artifact attempt authority ${attempt.attemptId}`); - } - authorities.set(attempt.attemptId, { node, ...attempt }); - } - } - return authorities; -} - -function plannedPrerequisiteAttempts( - node: PlannedGraphNodeDocument, - nodesById: ReadonlyMap -): PlannedAttemptAuthority[] { - return node.depends_on - .flatMap((dependencyId) => { - const dependency = nodesById.get(dependencyId); - if (dependency === undefined) { - throw new Error(`producer dependency is absent from the planned graph: ${dependencyId}`); - } - return plannedAttempts(dependency).map((attempt) => ({ node: dependency, ...attempt })); - }) - .sort((left, right) => left.attemptId.localeCompare(right.attemptId)); -} - -function authenticatedProducerAuthority( - layout: RunLayout, - nodes: readonly PlannedGraphNodeDocument[], - producer: PlannedGraphNodeDocument, - attemptId: string, - attemptIndex: number, - artifactRoot: string -): AuthenticatedProducerAuthority { - const markerRoot = path.resolve(layout.root, VERIFICATION_MARKER_DIRECTORY); - assertPathInside(layout.root, markerRoot, "artifact verification marker root"); - assertNoSymlinkComponents(layout.root, markerRoot, "artifact verification marker root"); - const markerPath = safeResolveInside(markerRoot, `${attemptId}.json`, "artifact verification marker"); - const markerBytes = readSinglyLinkedRegularFileSnapshotInside( - markerRoot, - markerPath, - MAX_AGGREGATION_AUTHORITY_BYTES, - "artifact verification marker" - ); - const parsed = parseStrictJsonBytes(markerBytes); - const shape = validateArtifactVerificationMarker(parsed); - if (!shape.ok) throw new Error(`artifact verification marker is invalid for ${attemptId}`); - const marker = parsed as ArtifactVerificationMarker; - assertArtifactVerificationMarkerSemantics(marker); - if ( - marker.schema_version !== ARTIFACT_VERIFICATION_SCHEMA_VERSION || - marker.attempt_id !== attemptId || - marker.node_id !== producer.logical_id || - marker.artifacts.length !== producer.outputs.length - ) { - throw new Error(`artifact verification marker identity changed for ${attemptId}`); - } - const markerArtifacts = new Map(marker.artifacts.map((artifact) => [artifact.path, artifact])); - if (markerArtifacts.size !== marker.artifacts.length) - throw new Error(`artifact verification marker repeats ${attemptId}`); - - const artifactManifestPath = safeResolveInside(artifactRoot, ARTIFACT_MANIFEST_FILE, "producer artifact manifest"); - const artifactManifestBytes = readSinglyLinkedRegularFileSnapshotInside( - artifactRoot, - artifactManifestPath, - MAX_AGGREGATION_AUTHORITY_BYTES, - "producer artifact manifest" - ); - const artifactManifestValue = parseStrictJsonBytes(artifactManifestBytes); - const artifactManifestShape = validateArtifactManifest(artifactManifestValue); - if (!artifactManifestShape.ok) throw new Error(`producer artifact manifest is invalid for ${attemptId}`); - const artifactManifest = artifactManifestValue as ArtifactManifest; - const provenanceMetadata = artifactManifest.provenance.metadata as { concrete_node_id?: string } | undefined; - if ( - artifactManifest.run_id !== layout.runId || - artifactManifest.node_id !== attemptId || - artifactManifest.producer_node_id !== attemptId || - artifactManifest.provenance.producer_node_id !== attemptId || - artifactManifest.provenance.run_id !== layout.runId || - artifactManifest.provenance.logical_node_id !== producer.logical_id || - artifactManifest.provenance.attempt_index !== attemptIndex || - provenanceMetadata?.concrete_node_id !== producer.id || - !isDeepStrictEqual(artifactManifest.output_contracts, producer.outputs) - ) { - throw new Error(`producer artifact manifest identity changed for ${attemptId}`); - } - authenticatePrerequisiteManifestChain(layout, nodes, artifactManifest, attemptId); - - const manifestFiles = new Map(artifactManifest.files.map((entry) => [entry.path, entry])); - const markerPublications = new Map(marker.publications.map((entry) => [entry.path, entry])); - if ( - manifestFiles.size !== artifactManifest.files.length || - markerPublications.size !== marker.publications.length || - manifestFiles.size !== markerPublications.size || - [...manifestFiles.keys()].some((relativePath) => !markerPublications.has(relativePath)) - ) { - throw new Error(`producer artifact manifest and verifier publication sets differ for ${attemptId}`); - } - const publications = new Map(); - for (const [relativePath, publication] of markerPublications) { - const manifestFile = manifestFiles.get(relativePath)!; - const absolutePath = safeResolveInside(artifactRoot, relativePath, "verified producer publication"); - const bytes = readSinglyLinkedRegularFileSnapshotInside( - artifactRoot, - absolutePath, - MAX_AGGREGATION_AUTHORITY_BYTES, - `verified producer publication ${relativePath}` - ); - const sha256 = digest(bytes); - if ( - publication.sha256 !== sha256 || - manifestFile.sha256 !== sha256 || - manifestFile.size_bytes !== bytes.byteLength - ) { - throw new Error(`verified producer publication changed ${attemptId}/${relativePath}`); - } - publications.set(relativePath, Object.freeze({ path: relativePath, absolutePath, bytes, sha256 })); - } - for (const output of producer.outputs) { - const artifact = markerArtifacts.get(output.path); - const expected = { - path: output.path, - contract: output.contract, - contract_digest: output.contract_digest, - ...(output.schema_file === undefined - ? {} - : { - schema_file: output.schema_file, - schema_id: output.schema_id, - schema_sha256: output.schema_sha256, - schema_bundle_sha256: output.schema_bundle_sha256, - validator_build: output.validator_build - }), - primary: output.primary - }; - if ( - artifact === undefined || - !isDeepStrictEqual( - { - path: artifact.path, - contract: artifact.contract, - contract_digest: artifact.contract_digest, - ...(artifact.schema_file === undefined ? {} : { schema_file: artifact.schema_file }), - ...(artifact.schema_id === undefined ? {} : { schema_id: artifact.schema_id }), - ...(artifact.schema_sha256 === undefined ? {} : { schema_sha256: artifact.schema_sha256 }), - ...(artifact.schema_bundle_sha256 === undefined - ? {} - : { schema_bundle_sha256: artifact.schema_bundle_sha256 }), - ...(artifact.validator_build === undefined ? {} : { validator_build: artifact.validator_build }), - primary: artifact.primary - }, - expected - ) - ) { - throw new Error(`artifact verification marker output binding changed ${attemptId}/${output.path}`); - } - const publication = publications.get(output.path); - if (publication === undefined || artifact.sha256 !== publication.sha256) { - throw new Error(`verified producer output publication changed ${attemptId}/${output.path}`); - } - } - return Object.freeze({ - marker, - artifactManifest, - artifactManifestBytes: Buffer.from(artifactManifestBytes), - publications - }); -} - -function authenticatePrerequisiteManifestChain( - layout: RunLayout, - nodes: readonly PlannedGraphNodeDocument[], - rootManifest: ArtifactManifest, - rootAttemptId: string -): void { - const nodesById = new Map(nodes.map((node) => [node.id, node])); - const attemptsById = plannedAttemptAuthorities(nodes); - const rootAuthority = attemptsById.get(rootAttemptId); - if (rootAuthority === undefined) { - throw new Error(`producer attempt is absent from the planned graph: ${rootAttemptId}`); - } - - const pending: Array<{ authority: PlannedAttemptAuthority; expectedSha256: string }> = []; - const scheduledDigests = new Map(); - const schedule = (authority: PlannedAttemptAuthority, expectedSha256: string): void => { - const previous = scheduledDigests.get(authority.attemptId); - if (previous !== undefined) { - if (previous !== expectedSha256) { - throw new Error(`prerequisite artifact manifest has conflicting sealed digests for ${authority.attemptId}`); - } - return; - } - if (scheduledDigests.size >= MAX_AGGREGATION_PREREQUISITE_MANIFESTS) { - throw new Error( - `prerequisite artifact manifest chain exceeds ${MAX_AGGREGATION_PREREQUISITE_MANIFESTS} manifests` - ); - } - scheduledDigests.set(authority.attemptId, expectedSha256); - pending.push({ authority, expectedSha256 }); - }; - - const rootExpectedPrerequisites = plannedPrerequisiteAttempts(rootAuthority.node, nodesById); - const rootExpectedById = new Map(rootExpectedPrerequisites.map((entry) => [entry.attemptId, entry])); - const rootActualIds = rootManifest.prerequisite_manifests.map((entry) => entry.node_id).sort(); - if (!isDeepStrictEqual(rootActualIds, [...rootExpectedById.keys()].sort())) { - throw new Error(`producer artifact manifest prerequisite set changed for ${rootAttemptId}`); - } - for (const prerequisite of rootManifest.prerequisite_manifests) { - schedule(rootExpectedById.get(prerequisite.node_id)!, prerequisite.sha256); - } - - let authenticatedManifestCount = 0; - let authenticatedEdgeCount = rootManifest.prerequisite_manifests.length; - let authenticatedBytes = 0; - if (authenticatedEdgeCount > MAX_AGGREGATION_PREREQUISITE_EDGES) { - throw new Error(`prerequisite artifact manifest chain exceeds ${MAX_AGGREGATION_PREREQUISITE_EDGES} edges`); - } - - while (pending.length > 0) { - const { authority, expectedSha256 } = pending.pop()!; - const remainingBytes = MAX_AGGREGATION_PREREQUISITE_BYTES - authenticatedBytes; - if (remainingBytes <= 0) { - throw new Error(`prerequisite artifact manifest chain exceeds ${MAX_AGGREGATION_PREREQUISITE_BYTES} bytes`); - } - const artifactRoot = safeResolveInside(layout.artifactsDir, authority.attemptId, "prerequisite artifact directory"); - const manifestPath = safeResolveInside(artifactRoot, ARTIFACT_MANIFEST_FILE, "prerequisite artifact manifest"); - const bytes = readSinglyLinkedRegularFileSnapshotInside( - artifactRoot, - manifestPath, - Math.min(MAX_AGGREGATION_AUTHORITY_BYTES, remainingBytes), - `prerequisite artifact manifest ${authority.attemptId}` - ); - authenticatedBytes += bytes.byteLength; - authenticatedManifestCount += 1; - const sha256 = digest(bytes); - if (sha256 !== expectedSha256) { - throw new Error(`prerequisite artifact manifest bytes changed for ${authority.attemptId}`); - } - const value = parseStrictJsonBytes(bytes); - const shape = validateArtifactManifest(value); - if (!shape.ok) throw new Error(`prerequisite artifact manifest is invalid for ${authority.attemptId}`); - const manifest = value as ArtifactManifest; - const provenanceMetadata = manifest.provenance.metadata as { concrete_node_id?: string } | undefined; - if ( - manifest.run_id !== layout.runId || - manifest.node_id !== authority.attemptId || - manifest.producer_node_id !== authority.attemptId || - manifest.provenance.run_id !== layout.runId || - manifest.provenance.producer_node_id !== authority.attemptId || - manifest.provenance.logical_node_id !== authority.node.logical_id || - !isDeepStrictEqual(manifest.output_contracts, authority.node.outputs) || - (authority.node.kind === "agentic" && - (manifest.provenance.attempt_index !== authority.attemptIndex || - provenanceMetadata?.concrete_node_id !== authority.node.id)) - ) { - throw new Error(`prerequisite artifact manifest identity changed for ${authority.attemptId}`); - } - - const expectedPrerequisites = plannedPrerequisiteAttempts(authority.node, nodesById); - const expectedById = new Map(expectedPrerequisites.map((entry) => [entry.attemptId, entry])); - const actualIds = manifest.prerequisite_manifests.map((entry) => entry.node_id).sort(); - if (!isDeepStrictEqual(actualIds, [...expectedById.keys()].sort())) { - throw new Error(`prerequisite artifact manifest prerequisite set changed for ${authority.attemptId}`); - } - authenticatedEdgeCount += manifest.prerequisite_manifests.length; - if (authenticatedEdgeCount > MAX_AGGREGATION_PREREQUISITE_EDGES) { - throw new Error(`prerequisite artifact manifest chain exceeds ${MAX_AGGREGATION_PREREQUISITE_EDGES} edges`); - } - for (const prerequisite of manifest.prerequisite_manifests) { - schedule(expectedById.get(prerequisite.node_id)!, prerequisite.sha256); - } - } - - if (authenticatedManifestCount !== scheduledDigests.size) { - throw new Error(`prerequisite artifact manifest chain authentication was incomplete for ${rootAttemptId}`); - } -} - -function authenticatedEntry( - kind: "generated-test" | "support-file", - entry: GeneratedTestEntry, - publications: ReadonlyMap -): SemanticAggregationSourceEntryContext { - const publication = publications.get(entry.path); - if ( - publication === undefined || - publication.bytes.length > MAX_GENERATED_TEST_COMPANION_BYTES || - publication.bytes.length !== entry.size_bytes || - publication.sha256 !== entry.sha256 - ) { - throw new Error(`verified generated-test companion publication changed ${entry.path}`); - } - return Object.freeze({ - kind, - sourceArtifactPath: publication.absolutePath, - sourceRelativePath: entry.path, - sizeBytes: entry.size_bytes, - sha256: entry.sha256, - bytes: Buffer.from(publication.bytes), - ...(entry.language === undefined ? {} : { language: entry.language }), - ...(entry.description === undefined ? {} : { description: entry.description }), - ...(entry.provenance === undefined ? {} : { provenance: Object.freeze({ ...entry.provenance }) }) - }); -} - -function digest(bytes: Uint8Array): string { - return crypto.createHash("sha256").update(bytes).digest("hex"); -} diff --git a/packages/runtime/src/artifact-gates.ts b/packages/runtime/src/artifact-gates.ts index 8bebe8a23..7633c41f8 100644 --- a/packages/runtime/src/artifact-gates.ts +++ b/packages/runtime/src/artifact-gates.ts @@ -5,7 +5,6 @@ import path from "node:path"; import { isDeepStrictEqual } from "node:util"; import { - artifactContractDefinition, artifactContractSchemaBinding, artifactSchemaDirectory, artifactSchemaRegistry, @@ -19,7 +18,6 @@ import { executeSemanticGate, getNodeArtifactDir, getNodeWorkspaceDir, - findingFuzzerBackendProvenance, INVARIANT_PINNED_SOURCE_REF, invariantPinnedSourceRefExists, IMPLEMENTED_PROPERTIES_SCHEMA_VERSION, @@ -78,7 +76,7 @@ import { type SemanticReviewStageContext, type WorkspacePatchManifest } from "@ultrafuzz/artifacts"; -import { parseProjectConfigToml } from "@ultrafuzz/config"; +import { invariantPropertyPrioritySelection, parseProjectConfigToml, type InvariantConfig } from "@ultrafuzz/config"; import { decodeHTML } from "entities"; import { fromMarkdown } from "mdast-util-from-markdown"; @@ -99,7 +97,6 @@ import { parseRuntimeDocumentBytes } from "./runtime-document-codec.js"; import type { PlannedGraph, PlannedGraphNode, RuntimeDiagnostic } from "./types.js"; import { topologyRuntimeBudgetForTimeout } from "./topology-runtime-budget.js"; import { diagnosticFromError } from "./utils.js"; -import { validateSeverityMatrixArtifact, type SeverityArtifactKind } from "./severity-matrix.js"; import { deriveWorkspacePatchGitFacts } from "./workspace-handoff.js"; import { loadFinalizedNodeOutputSnapshot, type VerifiedOutputArtifactSnapshot } from "./verified-output.js"; import { renderCoverageEvidenceMarkdownSection } from "./final-report-markdown.js"; @@ -319,10 +316,6 @@ function parseCurrentArtifactJson( return bytes === undefined ? undefined : parseStrictJsonBytes(bytes); } -export function verifyRequiredArtifactsForNode(layout: RunLayout, node: PlannedGraphNode): RequiredArtifactGate { - return verifyRequiredArtifactsForAttempt(layout, node, node.id); -} - function dynamicStrategyOutputTupleDiagnostics(layout: RunLayout, node: PlannedGraphNode): RuntimeDiagnostic[] { if (!node.outputs.some((output) => DYNAMIC_STRATEGY_OUTPUT_ROLE_CONTRACTS.has(output.contract))) return []; const invalidCounts = DYNAMIC_STRATEGY_OUTPUT_TUPLE_CONTRACTS.map((contract) => ({ @@ -347,7 +340,7 @@ export function verifyRequiredArtifactsForAttempt( layout: RunLayout, node: PlannedGraphNode, attemptId: string, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RequiredArtifactGate { const diagnostics: RuntimeDiagnostic[] = []; @@ -422,7 +415,6 @@ export function verifyRequiredArtifactsForAttempt( diagnostics.push(diagnosticFromError(error, "artifact-gates", "REQUIRED_ARTIFACT_INVALID")); } } - diagnostics.push(...verifySeverityMatrixArtifacts(artifactDir, node, authenticated)); try { diagnostics.push(...verifyInvariantEvidenceArtifacts(layout, artifactDir, node, attemptAuthority, authenticated)); } catch (error) { @@ -583,14 +575,12 @@ function verifyInvariantEvidenceArtifacts( layout: RunLayout, artifactDir: string, node: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RuntimeDiagnostic[] { const diagnostics: RuntimeDiagnostic[] = []; try { - diagnostics.push( - ...verifyInvariantLedgerProducerArtifacts(layout, artifactDir, node, attemptAuthority, authenticated) - ); + diagnostics.push(...verifyInvariantLedgerProducerArtifacts(layout, artifactDir, node, authenticated)); } catch (error) { diagnostics.push(diagnosticFromError(error, "invariant-ledger", "INVARIANT_EVIDENCE_READ_FAILED")); } @@ -608,7 +598,6 @@ function verifyInvariantLedgerProducerArtifacts( layout: RunLayout, artifactDir: string, node: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RuntimeDiagnostic[] { const diagnostics: RuntimeDiagnostic[] = []; @@ -638,79 +627,6 @@ function verifyInvariantLedgerProducerArtifacts( severity: "error" as const })) ); - if (parsed.value.entries.length === 0 && (parsed.value.scan_probes?.length ?? 0) > 0) { - verifyNoInvariantsJustification(parsed.value.no_invariants_justification, ledgerPath, diagnostics); - if (parsed.value.inventory_rows === undefined) { - diagnostics.push({ - code: "INVARIANT_LEDGER_INVENTORY_MISSING", - message: "An explicit no-evidence ledger must include an empty inventory_rows array", - severity: "error", - source: "invariant-ledger", - path: `${ledgerPath}#$.inventory_rows` - }); - } else if (parsed.value.inventory_rows.length > 0) { - diagnostics.push({ - code: "INVARIANT_LEDGER_INVENTORY_UNEXPECTED", - message: "An explicit no-evidence ledger must not contain inventory rows", - severity: "error", - source: "invariant-ledger", - path: `${ledgerPath}#$.inventory_rows` - }); - } - const discoveryWorkspace = path.join(layout.workspacesDir, path.basename(artifactDir)); - const sourceProofPath = invariantSourceProofPath(layout.root, path.basename(artifactDir)); - if (fs.existsSync(sourceProofPath)) { - readInvariantSourceProof(sourceProofPath, ledgerPath, ledgerBytes, diagnostics); - } else if (!fs.existsSync(discoveryWorkspace)) { - diagnostics.push({ - code: "INVARIANT_LEDGER_SOURCE_PROOF_MISSING", - message: "Invariant ledger source proof and discovery workspace are unavailable", - severity: "error", - source: "invariant-ledger", - path: ledgerPath - }); - } - // DECISION (issue #292), not a fact about probes: a probe path that names nothing is still - // accepted, because a probe records WHERE the agent looked and an optional file it did not - // find is a legitimate record. Containment stays the only property enforced here — probe - // `result` text has zero consumers, so it cannot be checked against anything. What the - // decision changed is the weight put on that text: it is no longer allowed to stand in for - // the claim "this target has no invariant". That claim now needs - // `no_invariants_justification` above. - for (const [probeIndex, probe] of (parsed.value.scan_probes ?? []).entries()) { - verifyInvariantProbePath(layout, discoveryWorkspace, probe.source_path, probeIndex, ledgerPath, diagnostics); - } - return diagnostics; - } - if (parsed.value.entries.length === 0) { - diagnostics.push({ - code: "INVARIANT_LEDGER_EMPTY", - message: "Project discovery invariant evidence ledger must contain at least one source entry", - severity: "error", - source: "invariant-ledger", - path: `${ledgerPath}#$.entries` - }); - return diagnostics; - } - if (parsed.value.inventory_rows === undefined) { - diagnostics.push({ - code: "INVARIANT_LEDGER_INVENTORY_MISSING", - message: "Project discovery invariant evidence ledger is missing structured inventory rows", - severity: "error", - source: "invariant-ledger", - path: `${ledgerPath}#$.inventory_rows` - }); - return diagnostics; - } - if (parsed.value.scan_probes === undefined) { - diagnostics.push({ - code: "INVARIANT_LEDGER_PROBES_MISSING", - message: "Project discovery invariant evidence ledger is missing scan probe results", - severity: "error", - source: "invariant-ledger", - path: `${ledgerPath}#$.scan_probes` - }); - } const discoveryWorkspace = path.join(layout.workspacesDir, path.basename(artifactDir)); const sourceProofPath = invariantSourceProofPath(layout.root, path.basename(artifactDir)); const sourceProofPresent = fs.existsSync(sourceProofPath); @@ -726,10 +642,14 @@ function verifyInvariantLedgerProducerArtifacts( path: ledgerPath }); } - // Same DECISION as the no-evidence branch above (issue #292): an absent probe path is accepted - // because a probe records where the agent looked, and containment is the only property that can - // be enforced when nothing consumes `result`. On this branch the ledger carries entries, and - // those remain byte-checked against the pinned source below. + // DECISION (issue #292), not a fact about probes: a probe path that names nothing is still + // accepted, because a probe records WHERE the agent looked and an optional file it did not + // find is a legitimate record. Containment stays the only property enforced here — probe + // `result` text has zero consumers, so it cannot be checked against anything. What the + // decision changed is the weight put on that text: it is no longer allowed to stand in for + // the claim "this target has no invariant". That claim needs `no_invariants_justification`, + // which the ledger schema requires when `entries` is empty. Ledger entries remain + // byte-checked against the pinned source below. for (const [probeIndex, probe] of (parsed.value.scan_probes ?? []).entries()) { verifyInvariantProbePath(layout, discoveryWorkspace, probe.source_path, probeIndex, ledgerPath, diagnostics); } @@ -747,7 +667,7 @@ function verifyCanonicalPropertiesProducerArtifacts( layout: RunLayout, artifactDir: string, node: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RuntimeDiagnostic[] { const diagnostics: RuntimeDiagnostic[] = []; @@ -851,49 +771,6 @@ function verifyCanonicalPropertiesProducerArtifacts( for (const [probeIndex, probe] of (ledger.value.scan_probes ?? []).entries()) { verifyInvariantProbePath(layout, discoveryWorkspace, probe.source_path, probeIndex, ledgerPath, diagnostics); } - if (ledger.value.entries.length === 0 && (ledger.value.scan_probes?.length ?? 0) > 0) { - verifyNoInvariantsJustification(ledger.value.no_invariants_justification, ledgerPath, diagnostics); - if (ledger.value.inventory_rows === undefined || ledger.value.inventory_rows.length > 0) { - diagnostics.push({ - code: "INVARIANT_LEDGER_INVENTORY_UNEXPECTED", - message: "An explicit no-evidence ledger must include an empty inventory_rows array", - severity: "error", - source: "invariant-ledger", - path: `${ledgerPath}#$.inventory_rows` - }); - } - for (const [propertyIndex, property] of catalog.value.properties.entries()) { - for (const [ledgerIndex, ledgerId] of (property.ledger_ids ?? []).entries()) { - diagnostics.push({ - code: "INVARIANT_LEDGER_REFERENCE_UNKNOWN", - message: `Canonical property ${JSON.stringify(property.id)} references ledger ID ${JSON.stringify(ledgerId)} but discovery recorded no invariant entries`, - severity: "error", - source: "invariant-ledger", - path: `${catalogPath}#$.properties[${propertyIndex}].ledger_ids[${ledgerIndex}]` - }); - } - } - return diagnostics; - } - if (ledger.value.entries.length === 0 || ledger.value.inventory_rows === undefined) { - diagnostics.push({ - code: "INVARIANT_LEDGER_INCOMPLETE", - message: "Property fan-in requires a non-empty invariant ledger with structured inventory rows", - severity: "error", - source: "invariant-ledger", - path: ledgerPath - }); - return diagnostics; - } - if (ledger.value.scan_probes === undefined) { - diagnostics.push({ - code: "INVARIANT_LEDGER_PROBES_MISSING", - message: "Property fan-in requires scan probe results from project discovery", - severity: "error", - source: "invariant-ledger", - path: `${ledgerPath}#$.scan_probes` - }); - } const ledgerIds = new Set(ledger.value.entries.map((entry) => entry.id)); const referenced = new Set(); @@ -943,80 +820,6 @@ function isSyntheticLedgerDependency(node: PlannedGraphNode): boolean { ); } -type DeclaredPropertyLensResolution = - { ok: true; output: NodeOutputContract } | { ok: false; diagnostic: RuntimeDiagnostic }; - -function resolveStateDeclaredPropertyLens( - state: RunState, - nodeId: string, - source: "property-fanin" | "property-provenance" -): DeclaredPropertyLensResolution { - const declarationPath = `state.nodes.${nodeId}.outputs`; - const outputs = (state.nodes[nodeId]?.outputs ?? []).filter((output) => output.contract === PROPERTY_LENS_CONTRACT); - if (outputs.length === 0) { - return { - ok: false, - diagnostic: { - code: "PROPERTY_LENS_DECLARATION_MISSING", - message: `State node ${JSON.stringify(nodeId)} must declare exactly one ${PROPERTY_LENS_CONTRACT} output; found none`, - severity: "error", - source, - path: declarationPath - } - }; - } - if (outputs.length !== 1) { - return { - ok: false, - diagnostic: { - code: "PROPERTY_LENS_DECLARATION_AMBIGUOUS", - message: `State node ${JSON.stringify(nodeId)} must declare exactly one ${PROPERTY_LENS_CONTRACT} output; found ${outputs.length}`, - severity: "error", - source, - path: declarationPath, - details: { declared_paths: outputs.map((output) => output.path) } - } - }; - } - - const output = outputs[0]!; - const definition = artifactContractDefinition(PROPERTY_LENS_CONTRACT); - const binding = artifactContractSchemaBinding(PROPERTY_LENS_CONTRACT); - const bindingMatches = - binding !== undefined && - output.contract_digest === definition.digest && - output.schema_file === binding.schema_file && - output.schema_id === binding.schema_id && - output.schema_sha256 === binding.schema_sha256 && - typeof output.schema_bundle_sha256 === "string" && - /^[0-9a-f]{64}$/u.test(output.schema_bundle_sha256) && - output.validator_build === binding.validator_build; - if (!bindingMatches) { - return { - ok: false, - diagnostic: { - code: "PROPERTY_LENS_SCHEMA_BINDING_INVALID", - message: `State node ${JSON.stringify(nodeId)} declares ${PROPERTY_LENS_CONTRACT} without its exact registered contract and schema binding`, - severity: "error", - source, - path: `${declarationPath}[${state.nodes[nodeId]?.outputs?.indexOf(output) ?? 0}]`, - details: { - expected: { contract_digest: definition.digest, ...binding }, - actual: { - contract_digest: output.contract_digest, - schema_file: output.schema_file, - schema_id: output.schema_id, - schema_sha256: output.schema_sha256, - schema_bundle_sha256: output.schema_bundle_sha256, - validator_build: output.validator_build - } - } - } - }; - } - return { ok: true, output }; -} - type FinalizedPropertyLensResolution = | { ok: true; @@ -1026,68 +829,6 @@ type FinalizedPropertyLensResolution = } | { ok: false; diagnostics: RuntimeDiagnostic[] }; -function loadFinalizedPropertyLens( - layout: RunLayout, - state: RunState, - nodeId: string, - source: "property-fanin" | "property-provenance" -): FinalizedPropertyLensResolution { - const declaration = resolveStateDeclaredPropertyLens(state, nodeId, source); - if (!declaration.ok) return { ok: false, diagnostics: [declaration.diagnostic] }; - const logicalNodeId = state.nodes[nodeId]?.logical_node_id ?? nodeId; - let authority: ReturnType; - try { - authority = loadFinalizedNodeOutputSnapshot({ - runRoot: layout.root, - logicalNodeId, - attemptId: nodeId - }); - } catch (error) { - return { - ok: false, - diagnostics: [ - { - code: "PROPERTY_LENS_AUTHORITY_INVALID", - message: `Property lens producer ${JSON.stringify(nodeId)} has no current finalized output authority: ${error instanceof Error ? error.message : String(error)}`, - severity: "error", - source, - path: `state.nodes.${nodeId}` - } - ] - }; - } - const artifacts = authority.outputs.filter((output) => output.contract === PROPERTY_LENS_CONTRACT); - if (artifacts.length !== 1 || artifacts[0]?.path !== declaration.output.path) { - return { - ok: false, - diagnostics: [ - { - code: "PROPERTY_LENS_AUTHORITY_INVALID", - message: `Finalized authority for ${JSON.stringify(nodeId)} does not bind its one declared ${PROPERTY_LENS_CONTRACT} output`, - severity: "error", - source, - path: `state.nodes.${nodeId}.outputs` - } - ] - }; - } - const artifact = artifacts[0]; - const typed = validateLensPropertiesSchema(artifact.value, artifact.absolute_path); - if (!typed.ok || typed.value === undefined) { - return { - ok: false, - diagnostics: typed.issues.map((issue) => ({ - code: issue.code, - message: issue.message, - severity: "error" as const, - source, - path: issue.path - })) - }; - } - return { ok: true, declaration: declaration.output, artifact, document: typed.value }; -} - type DirectArtifactDependency = { attemptId: string; node: PlannedGraphNode; @@ -1262,21 +1003,14 @@ function verifyLensReferenceExpectationPreservation( node: PlannedGraphNode, catalog: PropertiesArtifact, catalogPath: string, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): RuntimeDiagnostic[] { const diagnostics: RuntimeDiagnostic[] = []; const lensRows = new Map(); const lensPathsByDependency = new Map(); - const state = readRunState(layout); let dependencies: DirectArtifactDependency[]; try { - dependencies = - attemptAuthority === undefined - ? plannedDirectDependencyNodes(layout, node).map((dependency) => ({ - attemptId: dependency.id, - node: dependency - })) - : sealedDirectArtifactDependencies(layout, node, attemptAuthority); + dependencies = sealedDirectArtifactDependencies(layout, node, attemptAuthority); } catch (error) { return [diagnosticFromError(error, "property-fanin", "PROPERTY_LENS_AUTHORITY_INVALID")]; } @@ -1294,10 +1028,7 @@ function verifyLensReferenceExpectationPreservation( // Expanded graphs may give fan-in concrete dependencies such as // `property-specification-recon-0` and `property-specification-recon-1`. // Only that declared concrete dependency may satisfy the handoff. - const lens = - attemptAuthority === undefined - ? loadFinalizedPropertyLens(layout, state, dependencyId, "property-fanin") - : loadFinalizedTaskPropertyLens(layout, plannedDependency, "property-fanin"); + const lens = loadFinalizedTaskPropertyLens(layout, plannedDependency, "property-fanin"); if (!lens.ok) { diagnostics.push(...lens.diagnostics); continue; @@ -1510,40 +1241,6 @@ function verifyInvariantSourceEvidence( } } -/** - * Issue #292 verdict, recorded as a decision: REJECT SILENCE, ACCEPT EXPLICIT EMPTINESS. - * - * A ledger with no entries used to pass on the SHAPE of its emptiness alone — `inventory_rows: []` - * plus at least one scan probe — and that shape is free to fabricate: nothing anywhere reads - * `probe.result`, every consumer is a presence check, so invented probe text satisfied both evidence - * gates. "The agent searched and found nothing" was therefore unfalsifiable. - * - * A genuinely invariant-free target has to stay possible, so emptiness is not banned; it is made - * ATTRIBUTABLE. The ledger must state the claim in `no_invariants_justification`, which survives in - * the artifact and can be read against the target after the fact. - * - * Deliberately NOT the third option in the issue (mark the derived proof `partial`): that adds a - * state every consumer of the ledger has to learn, to describe a case that is already fully - * described by two existing ones. - */ -function verifyNoInvariantsJustification( - justification: string | undefined, - ledgerPath: string, - diagnostics: RuntimeDiagnostic[] -): void { - if (justification !== undefined) { - return; - } - diagnostics.push({ - code: "INVARIANT_LEDGER_NO_INVARIANTS_UNJUSTIFIED", - message: - "An invariant evidence ledger with no entries must record no_invariants_justification stating why the target carries no invariant", - severity: "error", - source: "invariant-ledger", - path: `${ledgerPath}#$.no_invariants_justification` - }); -} - function verifyInvariantProbePath( layout: RunLayout, workspacePath: string, @@ -1963,7 +1660,7 @@ function verifyRequiredArtifactShape( output: PlannedGraphNode["outputs"][number], node: PlannedGraphNode, attemptId: string, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RuntimeDiagnostic[] { const artifactBytes = @@ -2211,7 +1908,7 @@ function semanticGateContextForArtifact(input: { attemptId: string; output: PlannedGraphNode["outputs"][number]; schemaFilename: ArtifactSchemaFilename; - attemptAuthority?: ArtifactGateAttemptAuthority; + attemptAuthority: ArtifactGateAttemptAuthority; authenticated?: AuthenticatedArtifactGateSnapshots; document: unknown; }): SemanticGateContext { @@ -2220,7 +1917,6 @@ function semanticGateContextForArtifact(input: { rootDirectory: input.artifactDir, ...(input.authenticated === undefined ? {} : { files: input.authenticated.publications }) }, - plannedGraph: { node: input.node }, artifactIdentity: { runId: input.layout.runId, nodeId: input.node.logical_id ?? input.node.id, @@ -2237,8 +1933,13 @@ function semanticGateContextForArtifact(input: { input.schemaFilename === "aggregation-manifest.schema.json" ? authenticatedAggregationSemanticContext({ layout: input.layout, - node: input.node, - attemptId: input.attemptId + attemptId: input.attemptId, + producers: finalizedDeclaredContractProducers( + input.layout, + "ultrafuzz/generated-tests@3", + input.node, + input.attemptAuthority + ) }) : undefined; const propertyCampaignEvidence = @@ -2273,7 +1974,7 @@ function semanticArtifactSetForSchema(input: { attemptId: string; output: PlannedGraphNode["outputs"][number]; schemaFilename: ArtifactSchemaFilename; - attemptAuthority?: ArtifactGateAttemptAuthority; + attemptAuthority: ArtifactGateAttemptAuthority; authenticated?: AuthenticatedArtifactGateSnapshots; }): SemanticArtifactSetContext | undefined { if (input.schemaFilename === "campaign-summary.schema.json") { @@ -2401,7 +2102,7 @@ function semanticArtifactSetForSchema(input: { function semanticTriagedFindings( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): VerifiedOutputArtifactSnapshot | undefined { return finalizedSingletonAncestorOutput( layout, @@ -2438,7 +2139,7 @@ function semanticPropertyCampaignContext( layout: RunLayout, artifactDir: string, node: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): SemanticArtifactSetContext { const campaignPlan = semanticSiblingJsonArtifact( @@ -2488,12 +2189,12 @@ function authenticatedSemanticAttempt( artifactDir: string; node: PlannedGraphNode; attemptId: string; - attemptAuthority?: ArtifactGateAttemptAuthority; + attemptAuthority: ArtifactGateAttemptAuthority; authenticated?: AuthenticatedArtifactGateSnapshots; }, label: string ): { current: SemanticArtifactTaskDeclaration; authority: ArtifactGateAttemptAuthority } { - if (input.attemptAuthority === undefined || input.authenticated === undefined) { + if (input.authenticated === undefined) { throw new Error(`${label} requires sealed attempt declarations and authenticated current snapshots`); } if (input.attemptAuthority.task.attemptId !== input.attemptId) { @@ -2535,7 +2236,7 @@ function semanticDifferentialArtifacts(input: { attemptId: string; output: PlannedGraphNode["outputs"][number]; schemaFilename: ArtifactSchemaFilename; - attemptAuthority?: ArtifactGateAttemptAuthority; + attemptAuthority: ArtifactGateAttemptAuthority; authenticated?: AuthenticatedArtifactGateSnapshots; }): NonNullable { const { current, authority } = authenticatedSemanticAttempt(input, "differential semantic context"); @@ -2640,7 +2341,7 @@ function semanticDynamicStrategyArtifacts(input: { artifactDir: string; node: PlannedGraphNode; attemptId: string; - attemptAuthority?: ArtifactGateAttemptAuthority; + attemptAuthority: ArtifactGateAttemptAuthority; authenticated?: AuthenticatedArtifactGateSnapshots; }): NonNullable { const { authority } = authenticatedSemanticAttempt(input, "dynamic strategy semantic context"); @@ -2734,7 +2435,7 @@ function semanticReviewStageContext(input: { artifactDir: string; node: PlannedGraphNode; attemptId: string; - attemptAuthority?: ArtifactGateAttemptAuthority; + attemptAuthority: ArtifactGateAttemptAuthority; authenticated?: AuthenticatedArtifactGateSnapshots; }): SemanticReviewStageContext | undefined { authenticatedSemanticAttempt(input, "review stage semantic context"); @@ -2841,7 +2542,7 @@ function semanticReviewStageContext(input: { function semanticDedupedFindings( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, directOnly = true ): VerifiedOutputArtifactSnapshot | undefined { return finalizedSingletonAncestorOutput( @@ -2857,7 +2558,7 @@ function semanticDedupedFindings( function semanticFinalSeverityContext( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): { severityClassifiedFindings: unknown | null | undefined; dedupedFindings?: unknown | null; @@ -3008,7 +2709,7 @@ function semanticCampaignArtifacts( function semanticCanonicalPropertyCatalog( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): PropertiesArtifact | undefined { const pair = finalizedCanonicalPropertyPair(layout, consumer, attemptAuthority); if (pair !== undefined) return pair.value; @@ -3020,7 +2721,7 @@ function semanticCanonicalPropertyCatalog( function semanticImplementedPropertiesArtifact( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): { value: ImplementedPropertiesArtifact; path: string } | undefined { const artifact = finalizedSingletonAncestorOutput( layout, @@ -3042,7 +2743,7 @@ function semanticImplementedPropertiesArtifact( function semanticImplementedProperties( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): ImplementedPropertiesArtifact | undefined { const artifact = semanticImplementedPropertiesArtifact(layout, consumer, attemptAuthority); if (artifact !== undefined) return artifact.value; @@ -3055,7 +2756,7 @@ function semanticImplementedProperties( function semanticCampaignSummary( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): { value: unknown | null; path?: string } | undefined { const artifact = finalizedSingletonAncestorOutput( layout, @@ -3074,45 +2775,23 @@ function plannedContractProducerStatus( layout: RunLayout, consumer: PlannedGraphNode, contract: PlannedGraphNode["outputs"][number]["contract"], - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, includeOmitted = false -): "absent" | "present" | "unknown" { - if (attemptAuthority !== undefined) { - const { current, declarations } = semanticAttemptDeclarations(consumer, attemptAuthority); - assertRegularFileInside(layout.root, layout.graphPath, "planned contract producer authority"); - const graph = assertSealedPlannedGraph(readStrictRegisteredDocument(layout.graphPath, "planned-graph.schema.json")); - assertExactSealedAttemptAuthority(layout, graph, consumer, attemptAuthority); - const bindings = declaredAncestorOutputsByContract(current, declarations, contract); - if (bindings.length === 0) return "absent"; - if (includeOmitted) return "present"; - const sealedTasksByAttempt = new Map(attemptAuthority.tasks.map((task) => [task.attemptId, task] as const)); - const nodesById = new Map(graph.nodes.map((node) => [node.id, node] as const)); - const attemptIds = [...new Set(bindings.map((binding) => binding.attemptId))]; - return attemptIds.every((attemptId) => { - const sealedProducer = sealedTasksByAttempt.get(attemptId); - const concreteNodeId = sealedProducer?.concreteNodeId ?? attemptId; - const node = nodesById.get(concreteNodeId); - if (node === undefined) throw new Error(`planned contract producer is unavailable: ${attemptId}`); - return optionalDeclaredProducerWasNotAdmitted(graph, node, attemptId, sealedProducer, attemptAuthority); - }) - ? "absent" - : "present"; - } - if (!fs.existsSync(layout.graphPath)) return "unknown"; - let graph: ReturnType; - try { - assertRegularFileInside(layout.root, layout.graphPath, "planned graph semantic context"); - graph = assertSealedPlannedGraph(readStrictRegisteredDocument(layout.graphPath, "planned-graph.schema.json")); - } catch { - return "unknown"; - } - const ancestorIds = plannedAncestorIds(graph, consumer); - const producers = graph.nodes.filter( - (node) => ancestorIds.has(node.id) && node.outputs.some((output) => output.contract === contract) - ); - if (producers.length === 0) return "absent"; +): "absent" | "present" { + const { current, declarations } = semanticAttemptDeclarations(consumer, attemptAuthority); + assertRegularFileInside(layout.root, layout.graphPath, "planned contract producer authority"); + const graph = assertSealedPlannedGraph(readStrictRegisteredDocument(layout.graphPath, "planned-graph.schema.json")); + assertExactSealedAttemptAuthority(layout, graph, consumer, attemptAuthority); + const bindings = declaredAncestorOutputsByContract(current, declarations, contract); + if (bindings.length === 0) return "absent"; if (includeOmitted) return "present"; - return producers.every((node) => optionalDeclaredProducerWasNotAdmitted(graph, node, node.id, undefined, undefined)) + const sealedTasksByAttempt = new Map(attemptAuthority.tasks.map((task) => [task.attemptId, task] as const)); + const attemptIds = [...new Set(bindings.map((binding) => binding.attemptId))]; + return attemptIds.every((attemptId) => { + const sealedProducer = sealedTasksByAttempt.get(attemptId); + if (sealedProducer === undefined) throw new Error(`planned contract producer is unavailable: ${attemptId}`); + return optionalDeclaredProducerWasNotAdmitted(attemptId, sealedProducer, attemptAuthority); + }) ? "absent" : "present"; } @@ -3274,13 +2953,19 @@ function assertExactSealedAttemptAuthority( } const ancestorNodeIds = plannedAncestorIds(graph, plannedConsumer); - const expectedAncestorAttempts = graph.nodes - .filter((node) => ancestorNodeIds.has(node.id)) - .flatMap(plannedAttemptIdsForAuthority); - const expectedAncestorDirectories = expectedAncestorAttempts.map((attemptId) => - getNodeArtifactDir(layout, attemptId) - ); const actualAncestorDirectories = authority.task.dependencyArtifactDirs.map((directory) => path.resolve(directory)); + const sealedAncestorDirectories = new Set(actualAncestorDirectories); + // Dynamic lowering extends only a group's direct dependents, so a deeper + // descendant's sealed closure omits the generated attempts the lowered graph + // reaches through them. Gate contexts read ancestors through the sealed + // closure, exactly as the in-workflow verifier admitted them. + const expectedAncestorDirectories = graph.nodes + .filter((node) => ancestorNodeIds.has(node.id)) + .flatMap((node) => + plannedAttemptIdsForAuthority(node) + .map((attemptId) => getNodeArtifactDir(layout, attemptId)) + .filter((directory) => node.dynamic_generated === undefined || sealedAncestorDirectories.has(directory)) + ); assertExactStringSet( actualAncestorDirectories, expectedAncestorDirectories, @@ -3314,73 +2999,24 @@ function plannedAncestorIds( return ancestors; } -function plannedDirectDependencyNodes(layout: RunLayout, consumer: PlannedGraphNode): readonly PlannedGraphNode[] { - assertRegularFileInside(layout.root, layout.graphPath, "planned direct dependency authority"); - const graph = assertSealedPlannedGraph(readStrictRegisteredDocument(layout.graphPath, "planned-graph.schema.json")); - const plannedConsumers = graph.nodes.filter((node) => node.id === consumer.id); - if (plannedConsumers.length !== 1) { - throw new Error(`planned graph does not bind exact consumer ${JSON.stringify(consumer.id)}`); - } - const byId = new Map(graph.nodes.map((node) => [node.id, node] as const)); - const dependencies: PlannedGraphNode[] = []; - for (const dependencyId of plannedConsumers[0]!.depends_on) { - const dependency = byId.get(dependencyId); - if (dependency === undefined) { - throw new Error( - `planned consumer ${JSON.stringify(consumer.id)} names missing concrete dependency ${JSON.stringify(dependencyId)}` - ); - } - dependencies.push(dependency); - } - return dependencies; -} - function finalizedDeclaredContractProducers( layout: RunLayout, contract: PlannedGraphNode["outputs"][number]["contract"], consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, options: FinalizedDeclaredProducerOptions = {} ): FinalizedDeclaredProducer[] { assertRegularFileInside(layout.root, layout.graphPath, "planned graph finalized artifact authority"); const graph = assertSealedPlannedGraph(readStrictRegisteredDocument(layout.graphPath, "planned-graph.schema.json")); - let current: SemanticArtifactTaskDeclaration; - let declarations: SemanticArtifactTaskDeclaration[]; - const concreteNodeIdByAttempt = new Map(); + const { current, declarations } = semanticAttemptDeclarations(consumer, attemptAuthority); const sealedTaskByAttempt = new Map(); - if (attemptAuthority !== undefined) { - ({ current, declarations } = semanticAttemptDeclarations(consumer, attemptAuthority)); - for (const task of attemptAuthority.tasks) { - if (sealedTaskByAttempt.has(task.attemptId)) { - throw new Error(`sealed Smithers task set repeats attempt ${JSON.stringify(task.attemptId)}`); - } - sealedTaskByAttempt.set(task.attemptId, task); - concreteNodeIdByAttempt.set(task.attemptId, task.concreteNodeId); - } - } else { - const ancestorIds = plannedAncestorIds(graph, consumer); - const artifactDirectory = (node: PlannedGraphNode): string => - safeResolveInside(layout.root, node.artifact_dir, `planned artifact directory for ${node.id}`); - const ancestorDirectories = graph.nodes - .filter((node) => ancestorIds.has(node.id)) - .map((node) => artifactDirectory(node)); - declarations = graph.nodes.map((node) => ({ - attemptId: node.id, - logicalNodeId: node.logical_id, - artifactDir: artifactDirectory(node), - dependencies: node.depends_on, - dependencyArtifactDirs: node.id === consumer.id ? ancestorDirectories : [], - outputs: node.outputs - })); - current = declarations.find((declaration) => declaration.attemptId === consumer.id)!; - if (current === undefined) { - throw new Error(`planned graph does not bind exact consumer ${JSON.stringify(consumer.id)}`); + for (const task of attemptAuthority.tasks) { + if (sealedTaskByAttempt.has(task.attemptId)) { + throw new Error(`sealed Smithers task set repeats attempt ${JSON.stringify(task.attemptId)}`); } - for (const node of graph.nodes) concreteNodeIdByAttempt.set(node.id, node.id); - } - if (attemptAuthority !== undefined) { - assertExactSealedAttemptAuthority(layout, graph, consumer, attemptAuthority); + sealedTaskByAttempt.set(task.attemptId, task); } + assertExactSealedAttemptAuthority(layout, graph, consumer, attemptAuthority); const bindings = declaredAncestorOutputsByContract(current, declarations, contract, options); const bindingsByAttempt = new Map(); for (const binding of bindings) { @@ -3388,9 +3024,9 @@ function finalizedDeclaredContractProducers( } const nodesById = new Map(graph.nodes.map((node) => [node.id, node] as const)); return [...bindingsByAttempt.entries()].flatMap(([attemptId, declaredOutputs]) => { - const concreteNodeId = concreteNodeIdByAttempt.get(attemptId); - const node = concreteNodeId === undefined ? undefined : nodesById.get(concreteNodeId); - if (node === undefined) + const sealedTask = sealedTaskByAttempt.get(attemptId); + const node = sealedTask === undefined ? undefined : nodesById.get(sealedTask.concreteNodeId); + if (sealedTask === undefined || node === undefined) throw new Error(`declared semantic producer is absent from the planned graph: ${attemptId}`); if ( options.requiredSiblingContract !== undefined && @@ -3398,27 +3034,22 @@ function finalizedDeclaredContractProducers( ) { return []; } - const sealedTask = sealedTaskByAttempt.get(attemptId); if ( - attemptAuthority !== undefined && - (sealedTask === undefined || - node.kind !== "agentic" || - sealedTask.logicalNodeId !== node.logical_id || - (node.workflow !== undefined && !node.workflow.task_node_ids.includes(`node:${attemptId}`))) + node.kind !== "agentic" || + sealedTask.logicalNodeId !== node.logical_id || + (node.workflow !== undefined && !node.workflow.task_node_ids.includes(`node:${attemptId}`)) ) { throw new Error(`sealed semantic producer does not bind its planned attempt: ${attemptId}`); } - if (sealedTask !== undefined) { - const sealedOutputs = sealedTask.metadata.artifacts.outputs.filter((output) => output.contract === contract); - const plannedOutputs = node.outputs.filter((output) => output.contract === contract); - if ( - sealedOutputs.length !== plannedOutputs.length || - sealedOutputs.some((output) => !plannedOutputs.some((planned) => smithersOutputMatchesPlanned(output, planned))) - ) { - throw new Error(`sealed ${contract} declaration does not match the planned producer: ${attemptId}`); - } + const sealedOutputs = sealedTask.metadata.artifacts.outputs.filter((output) => output.contract === contract); + const plannedOutputs = node.outputs.filter((output) => output.contract === contract); + if ( + sealedOutputs.length !== plannedOutputs.length || + sealedOutputs.some((output) => !plannedOutputs.some((planned) => smithersOutputMatchesPlanned(output, planned))) + ) { + throw new Error(`sealed ${contract} declaration does not match the planned producer: ${attemptId}`); } - if (optionalDeclaredProducerWasNotAdmitted(graph, node, attemptId, sealedTask, attemptAuthority)) { + if (optionalDeclaredProducerWasNotAdmitted(attemptId, sealedTask, attemptAuthority)) { return []; } let outputAuthority: ReturnType; @@ -3445,27 +3076,19 @@ function finalizedDeclaredContractProducers( } function optionalDeclaredProducerWasNotAdmitted( - graph: ReturnType, - node: PlannedGraphNode, attemptId: string, - sealedProducer: SmithersTaskManifestTask | undefined, - authority: ArtifactGateAttemptAuthority | undefined + sealedProducer: SmithersTaskManifestTask, + authority: ArtifactGateAttemptAuthority ): boolean { - const optional = - sealedProducer !== undefined && authority !== undefined - ? (authority.task.optionalDependencyArtifactDirs ?? []).some( - (directory) => path.resolve(directory) === path.resolve(sealedProducer.artifactDir) - ) - : node.group !== undefined && graph.groups[node.group]?.defaults?.failure_policy === "continue"; + const optional = (authority.task.optionalDependencyArtifactDirs ?? []).some( + (directory) => path.resolve(directory) === path.resolve(sealedProducer.artifactDir) + ); if (!optional) return false; // Semantic consumption is authorized only by the consumer's verifier-bound // preparation decision. A marker that appears or disappears later cannot // enlarge or erase that immutable ancestor set. - return ( - authority === undefined || - !authenticatedDependencyAdmissionAttemptIds(authority.task, authority.admittedDependencyAttemptIds).includes( - attemptId - ) + return !authenticatedDependencyAdmissionAttemptIds(authority.task, authority.admittedDependencyAttemptIds).includes( + attemptId ); } @@ -3479,7 +3102,7 @@ interface FinalizedCanonicalPropertyPair { function finalizedCanonicalPropertyPair( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): FinalizedCanonicalPropertyPair | undefined { const bindings = finalizedDeclaredContractProducers( layout, @@ -3536,7 +3159,7 @@ function finalizedSingletonAncestorOutput( consumer: PlannedGraphNode, contract: PlannedGraphNode["outputs"][number]["contract"], label: string, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, options: { directOnly?: boolean; requiredSiblingContract?: PlannedGraphNode["outputs"][number]["contract"] } = {} ): VerifiedOutputArtifactSnapshot | undefined { const outputs = finalizedDeclaredContractProducers(layout, contract, consumer, attemptAuthority, options).flatMap( @@ -3552,9 +3175,8 @@ function finalizedSingletonAncestorOutput( function semanticPropertyLenses( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): SemanticPropertyLensContext[] | undefined { - const state = readRunState(layout); const lenses: SemanticPropertyLensContext[] = []; let producerCount = 0; // Discovery is the transitive root evidence authority for every property @@ -3597,23 +3219,7 @@ function semanticPropertyLenses( }); } - const directDependencies: DirectArtifactDependency[] = - attemptAuthority === undefined - ? plannedDirectDependencyNodes(layout, consumer).map((dependency) => ({ - attemptId: dependency.id, - node: dependency - })) - : sealedDirectArtifactDependencies(layout, consumer, attemptAuthority); - for (const dependency of directDependencies) { - const nodeId = dependency.attemptId; - const nodeState = state.nodes[nodeId]; - if ( - attemptAuthority === undefined && - nodeState?.logical_node_id !== undefined && - nodeState.logical_node_id !== dependency.node.logical_id - ) { - throw new Error(`property semantic dependency state does not bind planned node ${dependency.attemptId}`); - } + for (const dependency of sealedDirectArtifactDependencies(layout, consumer, attemptAuthority)) { const declaredLensCount = dependency.task?.metadata.artifacts.outputs.filter((output) => output.contract === PROPERTY_LENS_CONTRACT) .length ?? dependency.node.outputs.filter((output) => output.contract === PROPERTY_LENS_CONTRACT).length; @@ -3624,10 +3230,7 @@ function semanticPropertyLenses( ); } producerCount += 1; - const lens = - attemptAuthority === undefined - ? loadFinalizedPropertyLens(layout, state, nodeId, "property-fanin") - : loadFinalizedTaskPropertyLens(layout, dependency, "property-fanin"); + const lens = loadFinalizedTaskPropertyLens(layout, dependency, "property-fanin"); if (!lens.ok) { throw new Error( `property semantic lens authority is invalid for ${dependency.attemptId}: ${lens.diagnostics @@ -3691,49 +3294,40 @@ function readStrictRegisteredDocument(artifactPath: string, schemaFilename: Arti return document; } -function verifyRequiredArtifactSchemaBinding( +/** + * Validate one artifact against the schema content its planned output names. Verified reads use the + * same check, so a verified output stays valid for as long as its own schema snapshot exists. + */ +export function verifyRequiredArtifactSchemaBinding( layout: RunLayout, absolutePath: string, output: PlannedGraphNode["outputs"][number], artifactBytes: Uint8Array ): RuntimeDiagnostic[] { const current = artifactContractSchemaBinding(output.contract); - const actual = { + const planned = { schema_file: output.schema_file, schema_id: output.schema_id, schema_sha256: output.schema_sha256, schema_bundle_sha256: output.schema_bundle_sha256, validator_build: output.validator_build }; - if (current === undefined) { - if (Object.values(actual).every((value) => value === undefined)) return []; - return [schemaBindingMismatchDiagnostic(absolutePath, output, null, actual)]; - } - if ( - current.schema_file !== actual.schema_file || - current.schema_id !== actual.schema_id || - current.schema_sha256 !== actual.schema_sha256 || - current.validator_build !== actual.validator_build || - typeof actual.schema_bundle_sha256 !== "string" || - !/^[0-9a-f]{64}$/u.test(actual.schema_bundle_sha256) - ) { - return [schemaBindingMismatchDiagnostic(absolutePath, output, current, actual)]; + if (current === undefined && Object.values(planned).every((value) => value === undefined)) return []; + if (current === undefined || planned.schema_file === undefined || planned.schema_bundle_sha256 === undefined) { + return [schemaBindingMismatchDiagnostic(absolutePath, output, planned)]; } - const expected = { - schema_file: actual.schema_file, - schema_id: actual.schema_id, - schema_sha256: actual.schema_sha256, - schema_bundle_sha256: actual.schema_bundle_sha256, - validator_build: actual.validator_build - } as const; + // The planned binding names schema content. Its validator build only records which build planned + // the output and is never compared: validate with the installed schemas when they are the planned + // bundle, and otherwise with the bundle sealed into this run's execution snapshot, so a rebuild or + // upgrade does not change which schema an in-flight run's artifacts must satisfy (#921). let schemaPath: string; let schemaRegistry: ReturnType | undefined; try { - if (current.schema_bundle_sha256 === expected.schema_bundle_sha256) { - schemaPath = path.join(artifactSchemaDirectory(), current.schema_file); + if (current.schema_bundle_sha256 === planned.schema_bundle_sha256) { + schemaPath = safeResolveInside(artifactSchemaDirectory(), planned.schema_file, "planned artifact schema"); } else { - schemaPath = sealedArtifactSchemaPath(layout, current.schema_file); + schemaPath = sealedArtifactSchemaPath(layout, planned.schema_file); schemaRegistry = artifactSchemaRegistryFromDirectory(path.dirname(schemaPath)); } } catch (error) { @@ -3755,19 +3349,18 @@ function verifyRequiredArtifactSchemaBinding( ...(schemaRegistry === undefined ? {} : { schemaRegistry }) }); if ( - validation.schema?.id !== expected.schema_id || - validation.schema?.sha256 !== expected.schema_sha256 || - validation.schema?.bundle_sha256 !== expected.schema_bundle_sha256 || - validation.schema?.validator_build !== expected.validator_build + validation.schema?.id !== planned.schema_id || + validation.schema?.sha256 !== planned.schema_sha256 || + validation.schema?.bundle_sha256 !== planned.schema_bundle_sha256 ) { return [ { code: "ARTIFACT_VALIDATOR_IDENTITY_MISMATCH", - message: `Host validator identity for ${output.path} does not match the planned schema binding`, + message: `The schema available for ${output.path} does not match its planned schema binding`, severity: "error", source: "artifact-schema", path: absolutePath, - details: { contract: output.contract, expected, actual: validation.schema } + details: { contract: output.contract, expected: planned, actual: validation.schema } } ]; } @@ -3780,10 +3373,10 @@ function verifyRequiredArtifactSchemaBinding( path: `${absolutePath}${diagnostic.instancePath === undefined ? "" : `#${diagnostic.instancePath || "/"}`}`, details: { contract: output.contract, - schema_id: expected.schema_id, - schema_sha256: expected.schema_sha256, - schema_bundle_sha256: expected.schema_bundle_sha256, - validator_build: expected.validator_build, + schema_id: planned.schema_id, + schema_sha256: planned.schema_sha256, + schema_bundle_sha256: planned.schema_bundle_sha256, + validator_build: planned.validator_build, ...(diagnostic.schemaPath === undefined ? {} : { schema_path: diagnostic.schemaPath }), ...(diagnostic.keyword === undefined ? {} : { keyword: diagnostic.keyword }) } @@ -3793,16 +3386,15 @@ function verifyRequiredArtifactSchemaBinding( function schemaBindingMismatchDiagnostic( absolutePath: string, output: PlannedGraphNode["outputs"][number], - expected: ReturnType | null, - actual: Readonly> + planned: Readonly> ): RuntimeDiagnostic { return { code: "ARTIFACT_SCHEMA_BINDING_MISMATCH", - message: `Planned schema identity for ${output.path} does not match validator build ${expected?.validator_build ?? "unbound"}`, + message: `Planned schema binding for ${output.path} does not fit contract ${output.contract}`, severity: "error", source: "artifact-schema", path: absolutePath, - details: { contract: output.contract, expected, actual } + details: { contract: output.contract, planned } }; } @@ -3820,334 +3412,31 @@ function sealedArtifactSchemaPath(layout: RunLayout, schemaFile: string): string return schemaPath; } -function verifySeverityMatrixArtifacts( - artifactDir: string, - node: PlannedGraphNode, - authenticated?: AuthenticatedArtifactGateSnapshots -): RuntimeDiagnostic[] { - const artifact = severityArtifactForNode(node); - if (artifact === undefined) { - return []; - } - const artifactPath = safeResolveInside(artifactDir, artifact.path, "severity artifact output"); +/** The run's resolved `[invariants]` settings, read with the project config parser. */ +function resolvedInvariantConfig(layout: RunLayout): Partial | undefined { + if (!fs.existsSync(layout.resolvedConfigPath)) return undefined; try { - const document = parseCurrentArtifactJson(artifactDir, artifactPath, authenticated); - if (document === undefined) return []; - return validateSeverityMatrixArtifact({ - artifact: document, - artifactPath, - kind: artifact.kind - }); - } catch (error) { - return [diagnosticFromError(error, "severity-matrix", "SEVERITY_ARTIFACT_READ_FAILED")]; + const parsed = parseProjectConfigToml( + fs.readFileSync(layout.resolvedConfigPath, "utf8"), + layout.resolvedConfigPath + ); + return parsed.ok ? parsed.value.invariants : undefined; + } catch { + return undefined; } } -function severityArtifactForNode(node: PlannedGraphNode): { kind: SeverityArtifactKind; path: string } | undefined { - const logicalId = node.logical_id ?? node.id; - if (logicalId === "severity-classification") { - return { kind: "severity-classification", path: "severity-classified-findings.json" }; - } - const reportOutputs = node.outputs.filter((output) => output.contract === "ultrafuzz/report@3"); - if (reportOutputs.length === 1) { - return { kind: "final-report", path: reportOutputs[0]!.path }; - } - return undefined; -} - -const RECON_MAX_TEST_LIMIT = "18446744073709551615"; -const RECON_STATEFUL_SEQUENCE_LENGTH = 100; -const CAMPAIGN_HOST_FORCE_KILL_GRACE_SECONDS = 300; -const CAMPAIGN_DURATION_TOLERANCE_MS = 5_000; -const campaignTerminationReasons = new Set([ - "configured-timeout", - "test-limit", - "process-exit", - "launch-error", - "host-force-kill" -]); -const campaignOutcomes = new Set(["complete", "partial", "blocked"]); - -const currentCampaignRoleContracts = new Set([ - "ultrafuzz/invariant-campaign-plan@2", - "ultrafuzz/property-campaign@3", - "ultrafuzz/campaign-summary@2" -]); - -function hasCurrentCampaignOutputRole(node: PlannedGraphNode): boolean { - return node.outputs.some((output) => currentCampaignRoleContracts.has(output.contract)); -} - -function campaignTimeoutDiagnostic(code: string, message: string, pathValue: string): RuntimeDiagnostic { - return { - code, - message, - severity: "error", - source: "campaign-timeout-evidence", - path: pathValue - }; -} - -function declaredCampaignTimeoutArtifactPath( - artifactDir: string, - node: PlannedGraphNode, - contract: PlannedGraphNode["outputs"][number]["contract"], - label: string, - diagnostics: RuntimeDiagnostic[] -): string | undefined { - const outputs = node.outputs.filter((output) => output.contract === contract); - if (outputs.length !== 1) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_OUTPUT_DECLARATION_INVALID", - `Current campaign timeout evidence requires exactly one declared ${contract} ${label}; found ${outputs.length}`, - artifactDir - ) - ); - return undefined; - } - return safeResolveInside(artifactDir, outputs[0]!.path, `campaign timeout ${label}`); -} - -function positiveIntegerField( - record: Record, - field: string, - artifactPath: string, - diagnostics: RuntimeDiagnostic[] -): number | undefined { - const value = record[field]; - if (typeof value === "number" && Number.isSafeInteger(value) && value > 0) { - return value; - } - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - `${field} must be a positive safe integer`, - `${artifactPath}#$.${field}` - ) - ); - return undefined; -} - -function stringField( - record: Record, - field: string, - artifactPath: string, - diagnostics: RuntimeDiagnostic[] -): string | undefined { - const value = record[field]; - if (typeof value === "string" && value.length > 0) { - return value; - } - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - `${field} must be a non-empty string`, - `${artifactPath}#$.${field}` - ) - ); - return undefined; -} - -function timestampField( - record: Record, - field: string, - artifactPath: string, - diagnostics: RuntimeDiagnostic[] -): { text: string; milliseconds: number } | undefined { - const text = stringField(record, field, artifactPath, diagnostics); - if (text === undefined) return undefined; - const milliseconds = Date.parse(text); - if (Number.isFinite(milliseconds)) { - return { text, milliseconds }; - } - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - `${field} must be a valid timestamp`, - `${artifactPath}#$.${field}` - ) - ); - return undefined; -} - -function constrainedShellTokens(command: string): string[] | undefined { - const tokens: string[] = []; - let index = 0; - - const skipWhitespace = (): void => { - while (command[index] === " " || command[index] === "\t") index += 1; - }; - - const readWord = (redirectOperand = false): string | undefined => { - const start = index; - let quote: "'" | '"' | undefined; - let escaped = false; - while (index < command.length) { - const character = command[index]!; - if (escaped) { - escaped = false; - index += 1; - continue; - } - if (character === "\\" && quote !== "'") { - escaped = true; - index += 1; - continue; - } - if (character === "`" || (character === "$" && (redirectOperand || command[index + 1] === "("))) { - return undefined; - } - if (quote !== undefined) { - if (character === quote) quote = undefined; - index += 1; - continue; - } - if (character === "'" || character === '"') { - quote = character; - index += 1; - continue; - } - if (character === "\n" || character === "\r") return undefined; - if (character === " " || character === "\t") break; - if (character === ">" || character === "&") break; - if ( - character === "#" || - character === ";" || - character === "|" || - character === "<" || - character === "`" || - (character === "$" && command[index + 1] === "(") - ) { - return undefined; - } - index += 1; - } - if (quote !== undefined || escaped || index === start) return undefined; - return command.slice(start, index); - }; - - skipWhitespace(); - while (index < command.length) { - const duplication = command.slice(index).match(/^[0-9]+>&[0-9]+(?=$|[\s>])/u); - const redirection = duplication === null ? command.slice(index).match(/^(?:(?:[0-9]+)?(?:>>|>)|&>)/u) : null; - if (redirection !== null || duplication !== null) { - index += (redirection ?? duplication)![0].length; - if (redirection !== null) { - if (command[index] === "(") return undefined; - skipWhitespace(); - if (readWord(true) === undefined) return undefined; - } - skipWhitespace(); - continue; - } - const token = readWord(); - if (token === undefined) return undefined; - tokens.push(token); - if (/^[0-9]+$/u.test(token) && command.slice(index).startsWith("&>")) return undefined; - skipWhitespace(); - } - return tokens.length > 0 ? tokens : undefined; -} - -function exactReconCommandFlagValues(command: string, flag: "--timeout" | "--test-limit" | "--seq-len"): string[] { - const tokens = constrainedShellTokens(command); - if (tokens === undefined) return []; - const reconIndexes = tokens.flatMap((token, index) => - token === "recon" && tokens[index + 1] === "fuzz" ? [index] : [] - ); - if (reconIndexes.length !== 1) return []; - const argv = tokens.slice(reconIndexes[0]! + 2); - if (argv.includes("--")) return []; - const values: string[] = []; - for (let index = 0; index < argv.length; index += 1) { - const token = argv[index]!; - if (token === flag) { - values.push(argv[index + 1] ?? ""); - } else if (token.startsWith(`${flag}=`)) { - values.push(token.slice(flag.length + 1)); - } - } - return values; -} - -function hasExactHostTimeoutWrapper(command: string, configuredTimeoutSeconds: number): boolean { - const tokens = constrainedShellTokens(command); - if (tokens === undefined) return false; - const timeoutIndexes = tokens.flatMap((token, index) => (token === "timeout" ? [index] : [])); - if (timeoutIndexes.length !== 1 || tokens.includes("--foreground")) return false; - const timeoutIndex = timeoutIndexes[0]!; - const prefix = tokens.slice(0, timeoutIndex); - const assignmentStart = prefix[0] === "env" ? 1 : 0; - if ( - prefix.slice(assignmentStart).some((token) => !/^[A-Za-z_][A-Za-z0-9_]*=\S+$/u.test(token)) || - (prefix[0] === "env" && prefix.length === 1) - ) { - return false; - } - const reconIndex = tokens.indexOf("recon", timeoutIndex + 1); - if (reconIndex < 0 || tokens[reconIndex + 1] !== "fuzz") return false; - const wrapperArguments = tokens.slice(timeoutIndex + 1, reconIndex); - return ( - wrapperArguments.length === 4 && - wrapperArguments.at(-1) === `${configuredTimeoutSeconds}s` && - wrapperArguments.filter((argument) => argument === "--preserve-status").length === 1 && - wrapperArguments.filter((argument) => argument === "--signal=INT").length === 1 && - wrapperArguments.filter((argument) => argument === `--kill-after=${CAMPAIGN_HOST_FORCE_KILL_GRACE_SECONDS}s`) - .length === 1 - ); -} - -function readConfiguredInvariantFuzzerTimeoutSeconds(layout: RunLayout): number | undefined { - if (!fs.existsSync(layout.resolvedConfigPath)) return undefined; - try { - const parsed = parseProjectConfigToml( - fs.readFileSync(layout.resolvedConfigPath, "utf8"), - layout.resolvedConfigPath - ); - return parsed.ok ? parsed.value.invariants?.invariantTestingFuzzerTimeoutSeconds : undefined; - } catch { - return undefined; - } -} - -function configuredInvariantFuzzerTimeoutSeconds( - layout: RunLayout, - diagnostics: RuntimeDiagnostic[] -): number | undefined { - if (!fs.existsSync(layout.resolvedConfigPath)) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_CONFIG_MISSING", - "Current campaign timeout evidence requires the resolved configuration", - layout.resolvedConfigPath - ) - ); - return undefined; - } - const timeout = readConfiguredInvariantFuzzerTimeoutSeconds(layout); - if (timeout !== undefined) return timeout; - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_CONFIG_MISSING", - "Resolved configuration must declare invariants.invariant_testing_fuzzer_timeout", - layout.resolvedConfigPath - ) - ); - return undefined; -} - -function plannedCampaignTimeoutSeconds(node: PlannedGraphNode, attemptId: string): number | undefined { - const modelAttempt = node.model_fanout.find((model) => { - const plannedAttemptId = - model.attempt_id ?? - (node.model_fanout.length <= 1 - ? node.id - : `${node.id}__model_${model.model_index}__attempt_${model.attempt_index}`); - return plannedAttemptId === attemptId; - }); - const timeout = node.timeout_seconds ?? modelAttempt?.timeout_seconds; - return typeof timeout === "number" && Number.isSafeInteger(timeout) && timeout > 0 ? timeout : undefined; +function plannedCampaignTimeoutSeconds(node: PlannedGraphNode, attemptId: string): number | undefined { + const modelAttempt = node.model_fanout.find((model) => { + const plannedAttemptId = + model.attempt_id ?? + (node.model_fanout.length <= 1 + ? node.id + : `${node.id}__model_${model.model_index}__attempt_${model.attempt_index}`); + return plannedAttemptId === attemptId; + }); + const timeout = node.timeout_seconds ?? modelAttempt?.timeout_seconds; + return typeof timeout === "number" && Number.isSafeInteger(timeout) && timeout > 0 ? timeout : undefined; } function semanticPropertyCampaignTimeoutContext( @@ -4155,7 +3444,7 @@ function semanticPropertyCampaignTimeoutContext( node: PlannedGraphNode, attemptId: string ): SemanticPropertyCampaignTimeoutContext | undefined { - const configuredFuzzerTimeoutSeconds = readConfiguredInvariantFuzzerTimeoutSeconds(layout); + const configuredFuzzerTimeoutSeconds = resolvedInvariantConfig(layout)?.invariantTestingFuzzerTimeoutSeconds; const plannedTimeoutSeconds = plannedCampaignTimeoutSeconds(node, attemptId); if (configuredFuzzerTimeoutSeconds === undefined || plannedTimeoutSeconds === undefined) return undefined; return { @@ -4166,684 +3455,75 @@ function semanticPropertyCampaignTimeoutContext( }; } -function verifyCurrentCampaignTimeoutEvidence( +function verifyPropertyProvenanceArtifacts( layout: RunLayout, artifactDir: string, node: PlannedGraphNode, attemptId: string, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RuntimeDiagnostic[] { - if (!hasCurrentCampaignOutputRole(node)) return []; - + const isPropertyLens = node.outputs.some((output) => output.contract === "ultrafuzz/property-lens@2"); + const isFinalReport = node.outputs.some((output) => output.contract === "ultrafuzz/report@3"); + const isCoverageEvidence = node.outputs.some((output) => output.contract === "ultrafuzz/coverage-evidence@1"); + const isImplementation = node.outputs.some((output) => output.contract === "ultrafuzz/implemented-properties@3"); const diagnostics: RuntimeDiagnostic[] = []; - const planPath = declaredCampaignTimeoutArtifactPath( - artifactDir, - node, - "ultrafuzz/invariant-campaign-plan@2", - "plan", - diagnostics - ); - const resultPath = declaredCampaignTimeoutArtifactPath( - artifactDir, - node, - "ultrafuzz/property-campaign@3", - "result", - diagnostics - ); - const summaryPath = declaredCampaignTimeoutArtifactPath( - artifactDir, - node, - "ultrafuzz/campaign-summary@2", - "summary", - diagnostics - ); - const findingsPath = declaredCampaignTimeoutArtifactPath( - artifactDir, - node, - "ultrafuzz/findings@2", - "findings", - diagnostics - ); - if (planPath === undefined || resultPath === undefined || summaryPath === undefined || findingsPath === undefined) { - return diagnostics; - } - const planValue = parseCurrentArtifactJson(artifactDir, planPath, authenticated); - if (!isRecord(planValue) || planValue.schema_version !== "ultrafuzz.invariant-campaign-plan.v2") { - return [ - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - "The current invariant campaign-plan contract requires schema_version ultrafuzz.invariant-campaign-plan.v2", - `${planPath}#$.schema_version` - ) - ]; - } - const resultValue = parseCurrentArtifactJson(artifactDir, resultPath, authenticated); - const summaryValue = parseCurrentArtifactJson(artifactDir, summaryPath, authenticated); - if (!isRecord(resultValue)) { + if (isPropertyLens) { diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - "The current property-campaign result must be an object", - resultPath - ) + ...verifyLensReferenceExpectationAuthority(layout, artifactDir, node, attemptId, attemptAuthority, authenticated) ); } - if (!isRecord(summaryValue)) { + if (isFinalReport) { diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - "The current campaign summary must be an object", - summaryPath - ) + ...verifyFinalReportPropertyReferences(layout, artifactDir, node, attemptAuthority, authenticated) ); } - if (!isRecord(resultValue) || !isRecord(summaryValue)) return diagnostics; - - const configuredTimeoutSeconds = configuredInvariantFuzzerTimeoutSeconds(layout, diagnostics); - - const summarySequenceLength = positiveIntegerField(summaryValue, "sequence_length", summaryPath, diagnostics); - if (summarySequenceLength !== RECON_STATEFUL_SEQUENCE_LENGTH) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_SEQUENCE_LENGTH_MISMATCH", - `campaign summary sequence_length must be ${RECON_STATEFUL_SEQUENCE_LENGTH}`, - `${summaryPath}#$.sequence_length` - ) - ); + if (isCoverageEvidence) { + diagnostics.push(...verifyCoverageProductionInventory(layout, artifactDir, node, attemptId, authenticated)); + diagnostics.push(...verifyCoverageGoalEvidenceParity(artifactDir, node, authenticated)); } + if (!isImplementation) return diagnostics; - const planConfiguredTimeout = positiveIntegerField( - planValue, - "configured_fuzzer_timeout_seconds", - planPath, - diagnostics - ); - const configuredBudget = positiveIntegerField(planValue, "configured_budget_seconds", planPath, diagnostics); - const reconInternalTimeout = positiveIntegerField(planValue, "recon_internal_timeout_seconds", planPath, diagnostics); - const hostSoftTimeout = positiveIntegerField(planValue, "host_soft_timeout_seconds", planPath, diagnostics); - const forceKillGrace = positiveIntegerField(planValue, "host_force_kill_grace_seconds", planPath, diagnostics); - const finalizationReserve = positiveIntegerField( - planValue, - "artifact_finalization_reserve_seconds", - planPath, - diagnostics - ); - const requiredFinalizationReserve = positiveIntegerField( - planValue, - "finalization_reserve_seconds", - planPath, - diagnostics - ); - const plannedTimeoutSeconds = plannedCampaignTimeoutSeconds(node, attemptId); - if (plannedTimeoutSeconds === undefined) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_PLAN_BUDGET_MISSING", - "Current campaign timeout evidence requires the effective node, model-profile, or run-default timeout for this attempt in the sealed run graph", - `${layout.graphPath}#$.nodes.${node.id}.effective_timeout_seconds` - ) - ); - } else if (finalizationReserve !== undefined) { - const expectedReserve = topologyRuntimeBudgetForTimeout(plannedTimeoutSeconds * 1_000).finalizationReserveSeconds; - if (finalizationReserve !== expectedReserve) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_FINALIZATION_RESERVE_MISMATCH", - `artifact_finalization_reserve_seconds reports ${finalizationReserve}, but the sealed topology budget requires ${expectedReserve}`, - `${planPath}#$.artifact_finalization_reserve_seconds` - ) - ); - } - } - const reconTestLimit = stringField(planValue, "recon_test_limit", planPath, diagnostics); - const reconSequenceLength = positiveIntegerField(planValue, "recon_sequence_length", planPath, diagnostics); - const backendStartedAt = timestampField(planValue, "backend_started_at", planPath, diagnostics); - const fuzzingDeadline = timestampField(planValue, "fuzzing_deadline_utc", planPath, diagnostics); - const forceKillDeadline = timestampField(planValue, "force_kill_deadline_utc", planPath, diagnostics); - const finalArtifactDeadline = timestampField(planValue, "final_artifact_deadline_utc", planPath, diagnostics); - const requiredDeadline = timestampField(planValue, "deadline", planPath, diagnostics); - const planBackend = isRecord(planValue.backend) ? planValue.backend : undefined; - if (planBackend === undefined) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - "campaign plan backend must be an object", - `${planPath}#$.backend` - ) - ); + const catalog = readCanonicalPropertyCatalog(layout, node, attemptAuthority); + if (catalog.diagnostics.length > 0 || catalog.value === undefined) { + return [...diagnostics, ...catalog.diagnostics]; } - const planCommand = - planBackend === undefined - ? undefined - : stringField(planBackend, "exact_shell_escaped_command", `${planPath}#$.backend`, diagnostics); - for (const [field, value] of [ - ["configured_fuzzer_timeout_seconds", planConfiguredTimeout], - ["recon_internal_timeout_seconds", reconInternalTimeout], - ["host_soft_timeout_seconds", hostSoftTimeout] - ] as const) { - if (configuredTimeoutSeconds !== undefined && value !== undefined && value !== configuredTimeoutSeconds) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_CONFIG_MISMATCH", - `${field} reports ${value}, but resolved configuration requires ${configuredTimeoutSeconds}`, - `${planPath}#$.${field}` - ) - ); - } - } - if (reconTestLimit !== undefined && reconTestLimit !== RECON_MAX_TEST_LIMIT) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_TEST_LIMIT_MISMATCH", - `recon_test_limit must be ${RECON_MAX_TEST_LIMIT}`, - `${planPath}#$.recon_test_limit` - ) - ); - } - if (reconSequenceLength !== RECON_STATEFUL_SEQUENCE_LENGTH) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_SEQUENCE_LENGTH_MISMATCH", - `recon_sequence_length must be ${RECON_STATEFUL_SEQUENCE_LENGTH} for a stateful invariant campaign`, - `${planPath}#$.recon_sequence_length` - ) - ); - } - if (forceKillGrace !== undefined && forceKillGrace !== CAMPAIGN_HOST_FORCE_KILL_GRACE_SECONDS) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_HOST_GRACE_MISMATCH", - `host_force_kill_grace_seconds must be ${CAMPAIGN_HOST_FORCE_KILL_GRACE_SECONDS}`, - `${planPath}#$.host_force_kill_grace_seconds` - ) - ); - } - if ( - planConfiguredTimeout !== undefined && - forceKillGrace !== undefined && - finalizationReserve !== undefined && - configuredBudget !== undefined && - configuredBudget !== planConfiguredTimeout + forceKillGrace + finalizationReserve - ) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_PLAN_BUDGET_MISMATCH", - "configured_budget_seconds must equal the full fuzzer timeout plus host shutdown grace and artifact reserve", - `${planPath}#$.configured_budget_seconds` - ) - ); - } - if ( - requiredFinalizationReserve !== undefined && - finalizationReserve !== undefined && - requiredFinalizationReserve !== finalizationReserve - ) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_FINALIZATION_RESERVE_MISMATCH", - "finalization_reserve_seconds must equal artifact_finalization_reserve_seconds", - `${planPath}#$.finalization_reserve_seconds` - ) - ); - } - if ( - requiredDeadline !== undefined && - finalArtifactDeadline !== undefined && - requiredDeadline.milliseconds !== finalArtifactDeadline.milliseconds - ) { + const siblingImplementation = readDeclaredSiblingImplementedProperties(artifactDir, node, layout, authenticated); + diagnostics.push(...siblingImplementation.diagnostics); + if (siblingImplementation.value !== undefined && siblingImplementation.path !== undefined) { diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_DEADLINE_MISMATCH", - "The campaign plan deadline must equal final_artifact_deadline_utc", - `${planPath}#$.deadline` + ...verifyImplementationPropertyReferences( + layout, + artifactDir, + catalog.value, + node, + siblingImplementation.value, + siblingImplementation.path, + authenticated ) ); } + return diagnostics; +} - if ( - configuredTimeoutSeconds !== undefined && - forceKillGrace !== undefined && - finalizationReserve !== undefined && - backendStartedAt !== undefined && - fuzzingDeadline !== undefined && - forceKillDeadline !== undefined && - finalArtifactDeadline !== undefined - ) { - const expectedFuzzingDeadline = backendStartedAt.milliseconds + configuredTimeoutSeconds * 1_000; - const expectedForceKillDeadline = expectedFuzzingDeadline + forceKillGrace * 1_000; - const expectedFinalArtifactDeadline = expectedForceKillDeadline + finalizationReserve * 1_000; - for (const [field, actual, expected] of [ - ["fuzzing_deadline_utc", fuzzingDeadline.milliseconds, expectedFuzzingDeadline], - ["force_kill_deadline_utc", forceKillDeadline.milliseconds, expectedForceKillDeadline], - ["final_artifact_deadline_utc", finalArtifactDeadline.milliseconds, expectedFinalArtifactDeadline] - ] as const) { - if (actual !== expected) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_DEADLINE_MISMATCH", - `${field} does not match the configured timeout and reserve arithmetic`, - `${planPath}#$.${field}` - ) - ); +function verifyCoverageGoalEvidenceParity( + artifactDir: string, + node: PlannedGraphNode, + authenticated?: AuthenticatedArtifactGateSnapshots +): RuntimeDiagnostic[] { + const goalOutputs = node.outputs.filter((output) => output.contract === "ultrafuzz/coverage-goal@2"); + const evidenceOutputs = node.outputs.filter((output) => output.contract === "ultrafuzz/coverage-evidence@1"); + if (goalOutputs.length !== 1 || evidenceOutputs.length !== 1) { + return [ + { + code: "COVERAGE_GOAL_EVIDENCE_DECLARATION_AMBIGUOUS", + message: `Coverage producer must declare exactly one coverage goal and one coverage evidence output; found ${goalOutputs.length} goal and ${evidenceOutputs.length} evidence outputs`, + severity: "error", + source: "coverage-evidence", + path: node.id } - } - } - - const resultConfiguredTimeout = positiveIntegerField( - resultValue, - "configured_timeout_seconds", - resultPath, - diagnostics - ); - if ( - configuredTimeoutSeconds !== undefined && - resultConfiguredTimeout !== undefined && - resultConfiguredTimeout !== configuredTimeoutSeconds - ) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_CONFIG_MISMATCH", - `configured_timeout_seconds reports ${resultConfiguredTimeout}, but resolved configuration requires ${configuredTimeoutSeconds}`, - `${resultPath}#$.configured_timeout_seconds` - ) - ); - } - const resultCommand = stringField(resultValue, "exact_command", resultPath, diagnostics); - const resultSequenceLength = positiveIntegerField(resultValue, "sequence_length", resultPath, diagnostics); - if (resultSequenceLength !== RECON_STATEFUL_SEQUENCE_LENGTH) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_SEQUENCE_LENGTH_MISMATCH", - `sequence_length must be ${RECON_STATEFUL_SEQUENCE_LENGTH} for a stateful invariant campaign`, - `${resultPath}#$.sequence_length` - ) - ); - } - if (planCommand !== undefined && resultCommand !== undefined && planCommand !== resultCommand) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_COMMAND_MISMATCH", - "Backend result exact_command must equal the campaign plan command", - `${resultPath}#$.exact_command` - ) - ); - } - if (resultCommand !== undefined && configuredTimeoutSeconds !== undefined) { - const timeoutValues = exactReconCommandFlagValues(resultCommand, "--timeout"); - const testLimitValues = exactReconCommandFlagValues(resultCommand, "--test-limit"); - const sequenceLengthValues = exactReconCommandFlagValues(resultCommand, "--seq-len"); - if (timeoutValues.length !== 1 || timeoutValues[0] !== String(configuredTimeoutSeconds)) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_COMMAND_INVALID", - `Recon command must contain exactly one --timeout ${configuredTimeoutSeconds} flag`, - `${resultPath}#$.exact_command` - ) - ); - } - if (testLimitValues.length !== 1 || testLimitValues[0] !== RECON_MAX_TEST_LIMIT) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_COMMAND_INVALID", - `Recon command must contain exactly one --test-limit ${RECON_MAX_TEST_LIMIT} flag`, - `${resultPath}#$.exact_command` - ) - ); - } - if (sequenceLengthValues.length !== 1 || sequenceLengthValues[0] !== String(RECON_STATEFUL_SEQUENCE_LENGTH)) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - `Recon command must contain exactly one --seq-len ${RECON_STATEFUL_SEQUENCE_LENGTH} flag`, - `${resultPath}#$.exact_command` - ) - ); - } - if (!hasExactHostTimeoutWrapper(resultCommand, configuredTimeoutSeconds)) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_HOST_WRAPPER_INVALID", - `Recon command must use exactly one timeout --preserve-status --signal=INT --kill-after=${CAMPAIGN_HOST_FORCE_KILL_GRACE_SECONDS}s ${configuredTimeoutSeconds}s wrapper and must not use --foreground`, - `${resultPath}#$.exact_command` - ) - ); - } - } - - const startTimestamp = timestampField(resultValue, "start_timestamp", resultPath, diagnostics); - const endTimestamp = timestampField(resultValue, "end_timestamp", resultPath, diagnostics); - if ( - backendStartedAt !== undefined && - startTimestamp !== undefined && - backendStartedAt.milliseconds !== startTimestamp.milliseconds - ) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_START_MISMATCH", - "Backend result start_timestamp must equal campaign plan backend_started_at", - `${resultPath}#$.start_timestamp` - ) - ); - } - const terminationReason = stringField(resultValue, "termination_reason", resultPath, diagnostics); - if (terminationReason !== undefined && !campaignTerminationReasons.has(terminationReason)) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - `termination_reason must be one of ${[...campaignTerminationReasons].join(", ")}`, - `${resultPath}#$.termination_reason` - ) - ); - } - const campaignOutcome = stringField(resultValue, "campaign_outcome", resultPath, diagnostics); - if (campaignOutcome !== undefined && !campaignOutcomes.has(campaignOutcome)) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - "campaign_outcome must be complete, partial, or blocked", - `${resultPath}#$.campaign_outcome` - ) - ); - } - const usableResults = resultValue.usable_results; - if (typeof usableResults !== "boolean") { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - "usable_results must be a boolean", - `${resultPath}#$.usable_results` - ) - ); - } - - const executionValue = isRecord(resultValue.execution) ? resultValue.execution : undefined; - if (executionValue === undefined) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - "Backend result execution must be an object", - `${resultPath}#$.execution` - ) - ); - } else { - const executionCommand = stringField(executionValue, "command", `${resultPath}#$.execution`, diagnostics); - const executionStartedAt = timestampField(executionValue, "started_at", `${resultPath}#$.execution`, diagnostics); - const executionFinishedAt = timestampField(executionValue, "finished_at", `${resultPath}#$.execution`, diagnostics); - const executionDeadline = timestampField(executionValue, "deadline", `${resultPath}#$.execution`, diagnostics); - if (resultCommand !== undefined && executionCommand !== undefined && resultCommand !== executionCommand) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_COMMAND_MISMATCH", - "Backend result exact_command must equal execution.command", - `${resultPath}#$.execution.command` - ) - ); - } - if ( - startTimestamp !== undefined && - executionStartedAt !== undefined && - startTimestamp.milliseconds !== executionStartedAt.milliseconds - ) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_START_MISMATCH", - "Backend result start_timestamp must equal execution.started_at", - `${resultPath}#$.execution.started_at` - ) - ); - } - if ( - endTimestamp !== undefined && - executionFinishedAt !== undefined && - endTimestamp.milliseconds !== executionFinishedAt.milliseconds - ) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_END_MISMATCH", - "Backend result end_timestamp must equal execution.finished_at", - `${resultPath}#$.execution.finished_at` - ) - ); - } - if ( - finalArtifactDeadline !== undefined && - executionDeadline !== undefined && - finalArtifactDeadline.milliseconds !== executionDeadline.milliseconds - ) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_DEADLINE_MISMATCH", - "Backend execution.deadline must equal the plan final_artifact_deadline_utc", - `${resultPath}#$.execution.deadline` - ) - ); - } - if (typeof usableResults === "boolean" && executionValue.usable_results !== usableResults) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH", - "Backend usable_results must equal execution.usable_results", - `${resultPath}#$.execution.usable_results` - ) - ); - } - } - - if (startTimestamp !== undefined && endTimestamp !== undefined) { - const elapsedMs = endTimestamp.milliseconds - startTimestamp.milliseconds; - if (elapsedMs < 0) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_DURATION_MISMATCH", - "Backend end_timestamp cannot precede start_timestamp", - `${resultPath}#$.end_timestamp` - ) - ); - } else if (configuredTimeoutSeconds !== undefined) { - const endedEarly = elapsedMs + CAMPAIGN_DURATION_TOLERANCE_MS < configuredTimeoutSeconds * 1_000; - if (terminationReason === "configured-timeout" && endedEarly) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_DURATION_MISMATCH", - "A configured-timeout campaign must run for the configured fuzzer timeout", - `${resultPath}#$.end_timestamp` - ) - ); - } - if (endedEarly && usableResults === true && campaignOutcome !== "partial") { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH", - "A campaign that ends before the configured timeout with usable results must be partial", - `${resultPath}#$.campaign_outcome` - ) - ); - } - if ( - !endedEarly && - terminationReason === "configured-timeout" && - usableResults === true && - campaignOutcome !== "complete" - ) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH", - "A configured-timeout campaign with usable results must be complete", - `${resultPath}#$.campaign_outcome` - ) - ); - } - } - if ( - forceKillDeadline !== undefined && - endTimestamp.milliseconds > forceKillDeadline.milliseconds + CAMPAIGN_DURATION_TOLERANCE_MS - ) { - if (terminationReason !== "host-force-kill") { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_FORCE_KILL_MISMATCH", - "A backend ending after the host force-kill deadline must report termination_reason host-force-kill", - `${resultPath}#$.termination_reason` - ) - ); - } - if (usableResults === true && campaignOutcome !== "partial") { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH", - "A host-force-killed campaign with usable results must be partial", - `${resultPath}#$.campaign_outcome` - ) - ); - } - } - if ( - forceKillDeadline !== undefined && - terminationReason === "host-force-kill" && - endTimestamp.milliseconds + CAMPAIGN_DURATION_TOLERANCE_MS < forceKillDeadline.milliseconds - ) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_FORCE_KILL_MISMATCH", - "A host-force-kill termination cannot precede the host force-kill deadline", - `${resultPath}#$.termination_reason` - ) - ); - } - } - if (usableResults === false && campaignOutcome !== "blocked") { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH", - "A campaign without usable results must be blocked", - `${resultPath}#$.campaign_outcome` - ) - ); - } - if (campaignOutcome === "complete" && (terminationReason !== "configured-timeout" || usableResults !== true)) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH", - "A complete campaign must have usable results and termination_reason configured-timeout", - `${resultPath}#$.campaign_outcome` - ) - ); - } - if (usableResults === true && terminationReason !== "configured-timeout" && campaignOutcome !== "partial") { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH", - "A campaign with usable results and a non-configured terminal reason must be partial", - `${resultPath}#$.campaign_outcome` - ) - ); - } - if (usableResults === true && campaignOutcome === "blocked") { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH", - "A blocked campaign cannot report usable results", - `${resultPath}#$.campaign_outcome` - ) - ); - } - - const summaryOutcome = stringField(summaryValue, "outcome", summaryPath, diagnostics); - if (summaryOutcome !== undefined && campaignOutcome !== undefined && summaryOutcome !== campaignOutcome) { - diagnostics.push( - campaignTimeoutDiagnostic( - "CAMPAIGN_TIMEOUT_SUMMARY_MISMATCH", - "Campaign summary outcome must equal the backend result campaign_outcome", - `${summaryPath}#$.outcome` - ) - ); - } - return diagnostics; -} - -function verifyPropertyProvenanceArtifacts( - layout: RunLayout, - artifactDir: string, - node: PlannedGraphNode, - attemptId: string, - attemptAuthority?: ArtifactGateAttemptAuthority, - authenticated?: AuthenticatedArtifactGateSnapshots -): RuntimeDiagnostic[] { - const isPropertyLens = node.outputs.some((output) => output.contract === "ultrafuzz/property-lens@2"); - const isFinalReport = node.outputs.some((output) => output.contract === "ultrafuzz/report@3"); - const isCoverageEvidence = node.outputs.some((output) => output.contract === "ultrafuzz/coverage-evidence@1"); - const isImplementation = node.outputs.some((output) => output.contract === "ultrafuzz/implemented-properties@3"); - const isCampaign = node.outputs.some((output) => output.contract === "ultrafuzz/property-campaign@3"); - const diagnostics: RuntimeDiagnostic[] = []; - if (hasCurrentCampaignOutputRole(node)) { - diagnostics.push(...verifyCurrentCampaignTimeoutEvidence(layout, artifactDir, node, attemptId, authenticated)); - } - if (isPropertyLens) { - diagnostics.push( - ...verifyLensReferenceExpectationAuthority(layout, artifactDir, node, attemptId, attemptAuthority, authenticated) - ); - } - if (isFinalReport) { - diagnostics.push( - ...verifyFinalReportPropertyReferences(layout, artifactDir, node, attemptAuthority, authenticated) - ); - } - if (isCoverageEvidence) { - diagnostics.push(...verifyCoverageProductionInventory(layout, artifactDir, node, attemptId, authenticated)); - diagnostics.push(...verifyCoverageGoalEvidenceParity(artifactDir, node, authenticated)); - } - if (!isImplementation && !isCampaign) return diagnostics; - - const catalog = readCanonicalPropertyCatalog(layout, node, attemptAuthority); - if (catalog.diagnostics.length > 0 || catalog.value === undefined) { - return [...diagnostics, ...catalog.diagnostics]; - } - - const siblingImplementation = isImplementation - ? readDeclaredSiblingImplementedProperties(artifactDir, node, layout, authenticated) - : undefined; - if (isImplementation) { - diagnostics.push(...(siblingImplementation?.diagnostics ?? [])); - if (siblingImplementation?.value !== undefined && siblingImplementation.path !== undefined) { - diagnostics.push( - ...verifyImplementationPropertyReferences( - layout, - artifactDir, - catalog.value, - node, - siblingImplementation.value, - siblingImplementation.path, - authenticated - ) - ); - } - } - if (isCampaign) { - diagnostics.push( - ...verifyCampaignPropertyReferences(layout, artifactDir, catalog.value, node, attemptAuthority, authenticated) - ); - } - return diagnostics; -} - -function verifyCoverageGoalEvidenceParity( - artifactDir: string, - node: PlannedGraphNode, - authenticated?: AuthenticatedArtifactGateSnapshots -): RuntimeDiagnostic[] { - const goalOutputs = node.outputs.filter((output) => output.contract === "ultrafuzz/coverage-goal@2"); - const evidenceOutputs = node.outputs.filter((output) => output.contract === "ultrafuzz/coverage-evidence@1"); - if (goalOutputs.length !== 1 || evidenceOutputs.length !== 1) { - return [ - { - code: "COVERAGE_GOAL_EVIDENCE_DECLARATION_AMBIGUOUS", - message: `Coverage producer must declare exactly one coverage goal and one coverage evidence output; found ${goalOutputs.length} goal and ${evidenceOutputs.length} evidence outputs`, - severity: "error", - source: "coverage-evidence", - path: node.id - } - ]; + ]; } const goalPath = safeResolveInside(artifactDir, goalOutputs[0]!.path, "coverage goal output"); const evidencePath = safeResolveInside(artifactDir, evidenceOutputs[0]!.path, "coverage evidence output"); @@ -6145,9 +4825,8 @@ type ImplementedPropertiesRead = { }; /** - * Capture the implementation half of a mixed-role producer from its exact - * declared output. The immutable bytes are schema-validated before either the - * implementation gate or a sibling campaign join can consume the document. + * Capture the implementation producer's exact declared output. The immutable + * bytes are schema-validated before the implementation gate consumes them. */ function readDeclaredSiblingImplementedProperties( artifactDir: string, @@ -6228,937 +4907,421 @@ function verifyImplementationPropertyReferences( * A reference expectation is provenance, not free-form model metadata. A lens * may copy an ID only when the exact token is present in a declared structured * reference expectation catalog. Without such a catalog the field must be - * absent, keeping prose and code listings from becoming apparent authority. - */ -function verifyLensReferenceExpectationAuthority( - layout: RunLayout, - artifactDir: string, - node: PlannedGraphNode, - attemptId: string, - attemptAuthority?: ArtifactGateAttemptAuthority, - authenticated?: AuthenticatedArtifactGateSnapshots -): RuntimeDiagnostic[] { - let declaration: Pick; - if (attemptAuthority === undefined) { - const resolved = resolveStateDeclaredPropertyLens(readRunState(layout), attemptId, "property-provenance"); - if (!resolved.ok) return [resolved.diagnostic]; - declaration = resolved.output; - } else { - try { - assertRegularFileInside(layout.root, layout.graphPath, "current property lens attempt authority"); - const graph = assertSealedPlannedGraph( - readStrictRegisteredDocument(layout.graphPath, "planned-graph.schema.json") - ); - assertExactSealedAttemptAuthority(layout, graph, node, attemptAuthority); - } catch (error) { - return [diagnosticFromError(error, "property-provenance", "PROPERTY_LENS_AUTHORITY_INVALID")]; - } - const sealedOutputs = attemptAuthority.task.metadata.artifacts.outputs.filter( - (output) => output.contract === PROPERTY_LENS_CONTRACT - ); - const plannedOutputs = node.outputs.filter((output) => output.contract === PROPERTY_LENS_CONTRACT); - if ( - sealedOutputs.length !== 1 || - plannedOutputs.length !== 1 || - !smithersOutputMatchesPlanned(sealedOutputs[0]!, plannedOutputs[0]!) - ) { - return [ - { - code: "PROPERTY_LENS_SCHEMA_BINDING_INVALID", - message: `Current Smithers attempt ${JSON.stringify(attemptId)} must match its exact planned ${PROPERTY_LENS_CONTRACT} declaration`, - severity: "error", - source: "property-provenance", - path: `smithers.tasks.${attemptId}.metadata.artifacts.outputs` - } - ]; - } - declaration = sealedOutputs[0]!; - } - const lensPath = safeResolveInside(artifactDir, declaration.path, "property lens output"); - const lensDocument = parseCurrentArtifactJson(artifactDir, lensPath, authenticated); - if (lensDocument === undefined) { - return [ - { - code: "PROPERTY_LENS_MISSING", - message: `Declared property lens output ${JSON.stringify(declaration.path)} is unavailable`, - severity: "error", - source: "property-provenance", - path: lensPath - } - ]; - } - const lens = validateLensPropertiesSchema(lensDocument, lensPath); - if (!lens.ok || lens.value === undefined) { - return lens.issues.map((issue) => ({ - code: issue.code, - message: issue.message, - severity: "error", - source: "property-provenance", - path: issue.path - })); - } - - const supplied = readLensSuppliedExpectationIds(layout, node); - const suppliedExpectationIds = supplied.ids; - const diagnostics: RuntimeDiagnostic[] = [...supplied.diagnostics]; - for (const [propertyIndex, property] of lens.value.properties.entries()) { - if (property.reference_expectations !== undefined && property.reference_expectations.length === 0) { - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_OMISSION_REQUIRED", - message: `Property lens ${JSON.stringify(property.id)} must omit reference_expectations when it carries no authorized identifiers`, - severity: "error", - source: "property-provenance", - path: `${lensPath}#$.properties[${propertyIndex}].reference_expectations` - }); - } - if (!supplied.catalogSupplied && property.reference_expectations !== undefined) { - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_UNAUTHORIZED", - message: `Property lens ${JSON.stringify(property.id)} must omit reference_expectations because no structured catalog was supplied`, - severity: "error", - source: "property-provenance", - path: `${lensPath}#$.properties[${propertyIndex}].reference_expectations` - }); - continue; - } - for (const [expectationIndex, expectationId] of (property.reference_expectations ?? []).entries()) { - if (suppliedExpectationIds.has(expectationId)) continue; - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_UNAUTHORIZED", - message: `Property lens expectation ${JSON.stringify(expectationId)} is not present in a supplied pinned-reference catalog`, - severity: "error", - source: "property-provenance", - path: `${lensPath}#$.properties[${propertyIndex}].reference_expectations[${expectationIndex}]` - }); - } - } - return diagnostics; -} - -function readLensSuppliedExpectationIds( - layout: RunLayout, - node: PlannedGraphNode -): { ids: Set; catalogSupplied: boolean; diagnostics: RuntimeDiagnostic[] } { - const expectationIds = new Set(); - const diagnostics: RuntimeDiagnostic[] = []; - let catalogSupplied = false; - const state = readRunState(layout); - for (const dependencyId of node.depends_on) { - // Only declared, pinned-reference inputs can authorize provenance. In - // particular, an agentic setup/lens node or unrelated reference elsewhere - // in the run cannot authorize an ID for this lens. - const provenance = state.nodes[dependencyId]?.provenance; - if (provenance === undefined || !("origin" in provenance) || provenance.origin !== "pinned-reference") continue; - const dependencyDir = getNodeArtifactDir(layout, dependencyId); - const expectationPaths = [ - ...(declaresReferenceExpectationCatalog(state.nodes[dependencyId]?.outputs, "references/expectations.json") - ? ["references/expectations.json"] - : []) - ]; - if (expectationPaths.length === 0) continue; - catalogSupplied = true; - const metadata = provenance.reference_expectations; - if ( - !isRecord(metadata) || - metadata.source !== "operator-supplied" || - typeof metadata.sha256 !== "string" || - !/^[0-9a-f]{64}$/u.test(metadata.sha256) - ) { - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_PROVENANCE_INVALID", - message: `Pinned reference dependency ${JSON.stringify(dependencyId)} does not carry operator-supplied expectation provenance`, - severity: "error", - source: "property-provenance", - path: `state.nodes.${dependencyId}.provenance.reference_expectations` - }); - continue; - } - for (const expectationPath of expectationPaths) { - appendExpectationCatalog( - layout, - dependencyId, - path.join(dependencyDir, expectationPath), - expectationIds, - metadata.sha256, - diagnostics - ); - } - } - if (!catalogSupplied) { - // Issue #285: structured catalogs are the sole authority. Most benchmark - // runs intentionally supply none, so record that expectation-backed - // coverage is unavailable while requiring every lens row to omit the field. - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_CATALOG_ABSENT", - message: - "No pinned-reference dependency supplies a structured reference expectation catalog, so lens artifacts must omit reference_expectations", - severity: "warning", - source: "property-provenance", - path: `state.nodes.${node.id}.depends_on` - }); - } - return { ids: expectationIds, catalogSupplied, diagnostics }; -} - -function declaresReferenceExpectationCatalog( - outputs: ReadonlyArray<{ path: string; contract: string }> | undefined, - expectedPath: string -): boolean { - return ( - outputs?.some( - (output) => output.path === expectedPath && output.contract === "ultrafuzz/reference-expectations@2" - ) ?? false - ); -} - -function appendExpectationCatalog( - layout: RunLayout, - dependencyId: string, - catalogPath: string, - expectationIds: Set, - expectedDigest: string, - diagnostics: RuntimeDiagnostic[] -): void { - try { - const stat = fs.lstatSync(catalogPath); - if (!stat.isFile() || stat.isSymbolicLink()) { - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_TAMPERED", - message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} is not a regular file`, - severity: "error", - source: "property-provenance", - path: catalogPath - }); - return; - } - const catalogContents = readRegularFileSnapshot(catalogPath, MAX_ARTIFACT_SNAPSHOT_BYTES); - const actualDigest = sha256Bytes(catalogContents); - if (actualDigest !== expectedDigest) { - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_TAMPERED", - message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} does not match its recorded provenance digest`, - severity: "error", - source: "property-provenance", - path: catalogPath - }); - return; - } - const manifest = readArtifactManifest(layout, dependencyId); - const manifestEntry = manifest.files.find( - (file) => file.path === path.relative(getNodeArtifactDir(layout, dependencyId), catalogPath) - ); - if (manifestEntry?.sha256 !== expectedDigest) { - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_MANIFEST_MISMATCH", - message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} is not bound to its artifact manifest digest`, - severity: "error", - source: "property-provenance", - path: path.join(getNodeArtifactDir(layout, dependencyId), "artifact-manifest.json") - }); - return; - } - const parsed = validateReferenceExpectationsSchema(parseStrictJsonBytes(catalogContents), catalogPath); - if (!parsed.ok || parsed.value === undefined) { - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_TAMPERED", - message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} failed schema validation`, - severity: "error", - source: "property-provenance", - path: catalogPath - }); - return; - } - const uniqueness = executeSemanticGate("reference-expectation-id-uniqueness", { document: parsed.value }); - if (uniqueness.status === "failed" || uniqueness.status === "requires-context") { - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_TAMPERED", - message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} failed semantic validation`, - severity: "error", - source: "property-provenance", - path: catalogPath, - details: - uniqueness.status === "failed" - ? { gate: uniqueness.gate, issues: uniqueness.issues } - : { gate: uniqueness.gate, missing_context: uniqueness.missingContext } - }); - return; - } - for (const expectation of parsed.value.expectations) expectationIds.add(expectation.id); - } catch { - diagnostics.push({ - code: "PROPERTY_REFERENCE_EXPECTATION_MANIFEST_MISMATCH", - message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} has no verifiable artifact manifest`, - severity: "error", - source: "property-provenance", - path: catalogPath - }); - return; - } -} - -/** Validate the explicit priority selection required by the current contract. */ -function verifyImplementationSelectionCoverage( - catalog: PropertiesArtifact, - implementation: ImplementedPropertiesArtifact, - implementationPath: string, - layout: RunLayout -): RuntimeDiagnostic[] { - const selection = implementation.selection; - if (selection === undefined) { - return [ - { - code: "PROPERTY_IMPLEMENTATION_SELECTION_MISSING", - message: "Invariant implementation artifacts must declare selection metadata", - severity: "error", - source: "property-provenance", - path: `${implementationPath}#$.selection` - } - ]; - } - - const diagnostics: RuntimeDiagnostic[] = []; - const configuredSelection = readConfiguredInvariantPrioritySelection(layout); - if (configuredSelection === undefined) { - diagnostics.push({ - code: "PROPERTY_IMPLEMENTATION_CONFIG_MISSING", - message: - "Current invariant implementation coverage cannot be verified without resolved invariant priority configuration", - severity: "error", - source: "property-provenance", - path: layout.resolvedConfigPath - }); - } - const priorityOrder = ["high", "medium", "low"] as const; - const thresholdIndex = priorityOrder.indexOf(selection.priority_threshold); - const expectedPriorities = priorityOrder.slice(0, thresholdIndex + 1); - if ( - selection.priorities.length !== expectedPriorities.length || - selection.priorities.some((priority, index) => priority !== expectedPriorities[index]) - ) { - diagnostics.push({ - code: "PROPERTY_IMPLEMENTATION_SELECTION_INVALID", - message: `Implementation selection priorities must include exactly the priorities at or above ${JSON.stringify(selection.priority_threshold)}`, - severity: "error", - source: "property-provenance", - path: `${implementationPath}#$.selection.priorities` - }); - } - if ( - configuredSelection !== undefined && - (selection.priority_threshold !== configuredSelection.priority_threshold || - selection.priorities.length !== configuredSelection.priorities.length || - selection.priorities.some((priority, index) => priority !== configuredSelection.priorities[index])) - ) { - diagnostics.push({ - code: "PROPERTY_IMPLEMENTATION_SELECTION_CONFIG_MISMATCH", - message: "Implementation selection does not match the resolved invariant priority configuration", - severity: "error", - source: "property-provenance", - path: `${implementationPath}#$.selection` - }); - } - - const expectedIds = catalog.properties - .filter( - (property) => - selection.priorities.includes(property.priority) || - (configuredSelection?.reference_expectation_selection !== "priority" && - property.reference_expectations !== undefined && - property.reference_expectations.length > 0) - ) - .map((property) => property.id); - const selectedIds = new Set(selection.property_ids); - const expectedIdSet = new Set(expectedIds); - const missingSelectedIds = expectedIds.filter((propertyId) => !selectedIds.has(propertyId)); - const extraSelectedIds = selection.property_ids.filter((propertyId) => !expectedIdSet.has(propertyId)); - const selectionOrderMatches = - selection.property_ids.length === expectedIds.length && - selection.property_ids.every((propertyId, index) => propertyId === expectedIds[index]); - if (missingSelectedIds.length > 0 || extraSelectedIds.length > 0 || !selectionOrderMatches) { - diagnostics.push({ - code: "PROPERTY_IMPLEMENTATION_SELECTION_MISMATCH", - message: `Implementation selection must list every canonical property matching the configured priority and reference-expectation selection policy in catalog order (missing: ${JSON.stringify(missingSelectedIds)}, extra: ${JSON.stringify(extraSelectedIds)})`, - severity: "error", - source: "property-provenance", - path: `${implementationPath}#$.selection.property_ids` - }); - } - - const recordsById = new Map(implementation.properties.map((record) => [record.property_id, record])); - const recordIds = new Set(recordsById.keys()); - const missingRecords = expectedIds.filter((propertyId) => !recordIds.has(propertyId)); - const extraRecords = implementation.properties - .map((record) => record.property_id) - .filter((propertyId) => !expectedIdSet.has(propertyId)); - if (missingRecords.length > 0 || extraRecords.length > 0) { - diagnostics.push({ - code: "PROPERTY_IMPLEMENTATION_COVERAGE_INCOMPLETE", - message: `Implementation records must cover exactly the selected canonical properties (missing: ${JSON.stringify(missingRecords)}, extra: ${JSON.stringify(extraRecords)})`, - severity: "error", - source: "property-provenance", - path: `${implementationPath}#$.properties` - }); - } - - for (const [recordIndex, record] of implementation.properties.entries()) { - const canonical = catalog.properties.find((property) => property.id === record.property_id); - const expectedReferenceExpectations = canonical?.reference_expectations ?? []; - const actualReferenceExpectations = record.reference_expectations ?? []; - if (!sameStringSet(actualReferenceExpectations, expectedReferenceExpectations)) { - diagnostics.push({ - code: "PROPERTY_IMPLEMENTATION_REFERENCE_EXPECTATIONS_MISMATCH", - message: `Implementation record ${JSON.stringify(record.property_id)} must preserve the complete canonical reference expectation ID set`, - severity: "error", - source: "property-provenance", - path: `${implementationPath}#$.properties[${recordIndex}].reference_expectations` - }); - } - } - - for (const [recordIndex, record] of implementation.properties.entries()) { - if (!expectedIdSet.has(record.property_id) || record.status === "implemented") continue; - if (record.blocker !== undefined) continue; - diagnostics.push({ - code: "PROPERTY_IMPLEMENTATION_BLOCKER_MISSING", - message: `Selected property ${JSON.stringify(record.property_id)} is ${record.status} and must carry an actionable blocker with code, summary, and next_action`, - severity: "error", - source: "property-provenance", - path: `${implementationPath}#$.properties[${recordIndex}].blocker` - }); - } - return diagnostics; -} - -function readConfiguredInvariantPrioritySelection(layout: RunLayout): - | { - priority_threshold: "high" | "medium" | "low"; - priorities: ("high" | "medium" | "low")[]; - reference_expectation_selection?: "priority" | "mandatory"; - } - | undefined { - if (!fs.existsSync(layout.resolvedConfigPath)) return undefined; - const contents = fs.readFileSync(layout.resolvedConfigPath, "utf8"); - const match = /^\s*property_priority_threshold\s*=\s*["'](high|medium|low)["']\s*$/mu.exec(contents); - if (match === null) return undefined; - const priority_threshold = match[1] as "high" | "medium" | "low"; - const order = ["high", "medium", "low"] as const; - const policy = /^\s*reference_expectation_selection\s*=\s*["'](priority|mandatory)["']\s*$/mu.exec(contents); - return { - priority_threshold, - priorities: order.slice(0, order.indexOf(priority_threshold) + 1), - reference_expectation_selection: policy?.[1] as "priority" | "mandatory" | undefined - }; -} - -function verifyCampaignPropertyReferences( + * absent, keeping prose and code listings from becoming apparent authority. + */ +function verifyLensReferenceExpectationAuthority( layout: RunLayout, artifactDir: string, - catalog: PropertiesArtifact, node: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptId: string, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RuntimeDiagnostic[] { - const campaignOutputs = node.outputs.filter((output) => output.contract === "ultrafuzz/property-campaign@3"); - const findingOutputs = node.outputs.filter((output) => output.contract === "ultrafuzz/findings@2"); - const summaryOutputs = node.outputs.filter((output) => output.contract === "ultrafuzz/campaign-summary@2"); - if (findingOutputs.length !== 1 || summaryOutputs.length > 1) { + try { + assertRegularFileInside(layout.root, layout.graphPath, "current property lens attempt authority"); + const graph = assertSealedPlannedGraph(readStrictRegisteredDocument(layout.graphPath, "planned-graph.schema.json")); + assertExactSealedAttemptAuthority(layout, graph, node, attemptAuthority); + } catch (error) { + return [diagnosticFromError(error, "property-provenance", "PROPERTY_LENS_AUTHORITY_INVALID")]; + } + const sealedOutputs = attemptAuthority.task.metadata.artifacts.outputs.filter( + (output) => output.contract === PROPERTY_LENS_CONTRACT + ); + const plannedOutputs = node.outputs.filter((output) => output.contract === PROPERTY_LENS_CONTRACT); + const declaration = sealedOutputs.length === 1 ? sealedOutputs[0] : undefined; + const plannedOutput = plannedOutputs.length === 1 ? plannedOutputs[0] : undefined; + if ( + declaration === undefined || + plannedOutput === undefined || + !smithersOutputMatchesPlanned(declaration, plannedOutput) + ) { return [ { - code: "PROPERTY_CAMPAIGN_DECLARATION_AMBIGUOUS", - message: `Property campaign must declare one ultrafuzz/findings@2 output and at most one ultrafuzz/campaign-summary@2 output; found ${findingOutputs.length} and ${summaryOutputs.length}`, + code: "PROPERTY_LENS_SCHEMA_BINDING_INVALID", + message: `Current Smithers attempt ${JSON.stringify(attemptId)} must match its exact planned ${PROPERTY_LENS_CONTRACT} declaration`, severity: "error", source: "property-provenance", - path: layout.graphPath + path: `smithers.tasks.${attemptId}.metadata.artifacts.outputs` } ]; } - const findingsPath = safeResolveInside(artifactDir, findingOutputs[0]!.path, "campaign finding output"); - const campaignPaths = campaignOutputs.map((output) => - safeResolveInside(artifactDir, output.path, "property campaign output") - ); - const summaryPath = - summaryOutputs.length === 0 - ? undefined - : safeResolveInside(artifactDir, summaryOutputs[0]!.path, "campaign summary output"); - const campaignDocuments = campaignPaths.map((campaignPath) => - parseCurrentArtifactJson(artifactDir, campaignPath, authenticated) - ); - const findingsDocument = parseCurrentArtifactJson(artifactDir, findingsPath, authenticated); - if (campaignDocuments.some((document) => document === undefined) || findingsDocument === undefined) return []; - const campaigns = campaignPaths.flatMap((campaignPath) => { - const campaign = validatePropertyCampaignSchema( - campaignDocuments[campaignPaths.indexOf(campaignPath)], - campaignPath - ); - return campaign.ok && campaign.value !== undefined ? [{ path: campaignPath, value: campaign.value }] : []; - }); - const findings = validateFindingsSchema(findingsDocument, findingsPath); - if (!findings.ok || findings.value === undefined) { - return []; + const lensPath = safeResolveInside(artifactDir, declaration.path, "property lens output"); + const lensDocument = parseCurrentArtifactJson(artifactDir, lensPath, authenticated); + if (lensDocument === undefined) { + return [ + { + code: "PROPERTY_LENS_MISSING", + message: `Declared property lens output ${JSON.stringify(declaration.path)} is unavailable`, + severity: "error", + source: "property-provenance", + path: lensPath + } + ]; } - const validatedFindings = findings.value; - const campaignValues = campaigns.map((campaign) => campaign.value); - const summaryDiagnostics = - campaigns.length === campaignPaths.length - ? campaignSummaryFailureCountDiagnostics( - artifactDir, - summaryPath, - campaignValues, - validatedFindings, - authenticated - ) - : []; - const backendDiagnostics = - campaigns.length === campaignPaths.length - ? campaignFindingFuzzerBackendDiagnostics(campaignValues, validatedFindings, findingsPath) - : []; - const partitionDiagnostics = - campaigns.length === campaignPaths.length - ? campaignFailurePartitionDiagnostics(campaigns, validatedFindings, findingsPath) - : []; - const implementation = readImplementedProperties(layout, node, attemptAuthority); - if (implementation.diagnostics.length > 0 || implementation.value === undefined) { - return [...summaryDiagnostics, ...backendDiagnostics, ...partitionDiagnostics, ...implementation.diagnostics]; + const lens = validateLensPropertiesSchema(lensDocument, lensPath); + if (!lens.ok || lens.value === undefined) { + return lens.issues.map((issue) => ({ + code: issue.code, + message: issue.message, + severity: "error", + source: "property-provenance", + path: issue.path + })); } - const candidateFindingIds = new Set( - campaigns.flatMap((campaign) => campaign.value.failures.map((failure) => failure.id)) - ); - - const references: PropertyReferenceInput[] = campaigns.flatMap((campaign) => - campaign.value.failures.flatMap((failure, index) => - failure.property_ids === undefined - ? [] - : failure.property_ids.map((propertyId, propertyIndex) => ({ - propertyIds: [propertyId], - path: `${campaign.path}#$.failures[${index}].property_ids[${propertyIndex}]` - })) - ) - ); - references.push(...findingPropertyReferences(validatedFindings, findingsPath)); - const diagnostics = [ - ...summaryDiagnostics, - ...backendDiagnostics, - ...partitionDiagnostics, - ...propertyReferenceDiagnostics(catalog, references), - ...campaigns.flatMap((campaign) => - campaignFindingReferenceDiagnostics( - campaign.value.failures, - validatedFindings, - candidateFindingIds, - campaign.path, - findingsPath - ) - ), - ...danglingCampaignFindingDiagnostics( - new Set(campaigns.flatMap((campaign) => campaign.value.failures.map((failure) => failure.id))), - validatedFindings, - findingsPath - ), - ...unobservedFindingPropertyDiagnostics( - new Set( - campaigns.flatMap((campaign) => campaign.value.failures.flatMap((failure) => failure.property_ids ?? [])) - ), - validatedFindings, - findingsPath - ) - ]; - const implementedIds = new Set( - implementation.value.properties - .filter((record) => record.status === "implemented") - .map((record) => record.property_id) - ); - for (const reference of references) { - for (const propertyId of reference.propertyIds) { - if (!implementedIds.has(propertyId)) { - diagnostics.push({ - code: "PROPERTY_IMPLEMENTATION_REFERENCE_INVALID", - message: `Property ${JSON.stringify(propertyId)} is outside the exact implemented-property set`, - severity: "error", - source: "property-provenance", - path: reference.path - }); - } + const supplied = readLensSuppliedExpectationIds(layout, node); + const suppliedExpectationIds = supplied.ids; + const diagnostics: RuntimeDiagnostic[] = [...supplied.diagnostics]; + for (const [propertyIndex, property] of lens.value.properties.entries()) { + if (property.reference_expectations !== undefined && property.reference_expectations.length === 0) { + diagnostics.push({ + code: "PROPERTY_REFERENCE_EXPECTATION_OMISSION_REQUIRED", + message: `Property lens ${JSON.stringify(property.id)} must omit reference_expectations when it carries no authorized identifiers`, + severity: "error", + source: "property-provenance", + path: `${lensPath}#$.properties[${propertyIndex}].reference_expectations` + }); + } + if (!supplied.catalogSupplied && property.reference_expectations !== undefined) { + diagnostics.push({ + code: "PROPERTY_REFERENCE_EXPECTATION_UNAUTHORIZED", + message: `Property lens ${JSON.stringify(property.id)} must omit reference_expectations because no structured catalog was supplied`, + severity: "error", + source: "property-provenance", + path: `${lensPath}#$.properties[${propertyIndex}].reference_expectations` + }); + continue; + } + for (const [expectationIndex, expectationId] of (property.reference_expectations ?? []).entries()) { + if (suppliedExpectationIds.has(expectationId)) continue; + diagnostics.push({ + code: "PROPERTY_REFERENCE_EXPECTATION_UNAUTHORIZED", + message: `Property lens expectation ${JSON.stringify(expectationId)} is not present in a supplied pinned-reference catalog`, + severity: "error", + source: "property-provenance", + path: `${lensPath}#$.properties[${propertyIndex}].reference_expectations[${expectationIndex}]` + }); } } return diagnostics; } -interface CampaignFailureReference { - fuzzer_backend: string; - failure_id: string; - raw_result_ref: string; -} - -interface PartitionedCampaignFailure { - key: string; - id: string; - fuzzerBackend?: string; - rawResultRef: string; - propertyIds: readonly string[]; - path: string; -} - -/** Prove the producer-owned failure-to-finding partition for every campaign. */ -function campaignFailurePartitionDiagnostics( - campaigns: readonly { path: string; value: PropertyCampaignArtifact }[], - findings: readonly Readonly>[], - findingsPath: string -): RuntimeDiagnostic[] { - const failures: PartitionedCampaignFailure[] = campaigns.flatMap((campaign, campaignIndex) => - campaign.value.failures.map((failure, failureIndex) => ({ - key: `${campaignIndex}\u0000${failureIndex}`, - id: failure.id, - ...(campaign.value.fuzzer_backend === undefined ? {} : { fuzzerBackend: campaign.value.fuzzer_backend }), - rawResultRef: path.basename(campaign.path), - propertyIds: failure.property_ids ?? [], - path: `${campaign.path}#$.failures[${failureIndex}].id` - })) - ); - const failuresById = new Map(); - const failuresByBackendAndId = new Map(); - for (const failure of failures) { - const byId = failuresById.get(failure.id) ?? []; - byId.push(failure); - failuresById.set(failure.id, byId); - if (failure.fuzzerBackend !== undefined) { - const qualifiedKey = campaignBackendFailureKey(failure.fuzzerBackend, failure.id); - const qualified = failuresByBackendAndId.get(qualifiedKey) ?? []; - qualified.push(failure); - failuresByBackendAndId.set(qualifiedKey, qualified); - } - } - +function readLensSuppliedExpectationIds( + layout: RunLayout, + node: PlannedGraphNode +): { ids: Set; catalogSupplied: boolean; diagnostics: RuntimeDiagnostic[] } { + const expectationIds = new Set(); const diagnostics: RuntimeDiagnostic[] = []; - const claimedBy = new Map(); - for (const [findingIndex, finding] of findings.entries()) { - const propertyIds = Array.isArray(finding.property_ids) - ? finding.property_ids.filter((propertyId): propertyId is string => typeof propertyId === "string") - : []; - const hasContributions = Object.prototype.hasOwnProperty.call(finding, "contributing_backend_failures"); - const hasDeduplication = Object.prototype.hasOwnProperty.call(finding, "deduplication"); - const mustAccount = propertyIds.length > 0 || hasContributions || hasDeduplication; - if (!mustAccount) continue; - - const contributionPath = `${findingsPath}#$[${findingIndex}].contributing_backend_failures`; - const deduplicationPath = `${findingsPath}#$[${findingIndex}].deduplication`; - if (!hasContributions) { + let catalogSupplied = false; + const state = readRunState(layout); + for (const dependencyId of node.depends_on) { + // Only declared, pinned-reference inputs can authorize provenance. In + // particular, an agentic setup/lens node or unrelated reference elsewhere + // in the run cannot authorize an ID for this lens. + const provenance = state.nodes[dependencyId]?.provenance; + if (provenance === undefined || !("origin" in provenance) || provenance.origin !== "pinned-reference") continue; + const dependencyDir = getNodeArtifactDir(layout, dependencyId); + const expectationPaths = [ + ...(declaresReferenceExpectationCatalog(state.nodes[dependencyId]?.outputs, "references/expectations.json") + ? ["references/expectations.json"] + : []) + ]; + if (expectationPaths.length === 0) continue; + catalogSupplied = true; + const metadata = provenance.reference_expectations; + if ( + !isRecord(metadata) || + metadata.source !== "operator-supplied" || + typeof metadata.sha256 !== "string" || + !/^[0-9a-f]{64}$/u.test(metadata.sha256) + ) { diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_REQUIRED", - message: `Finding ${JSON.stringify(finding.id)} must declare contributing_backend_failures`, + code: "PROPERTY_REFERENCE_EXPECTATION_PROVENANCE_INVALID", + message: `Pinned reference dependency ${JSON.stringify(dependencyId)} does not carry operator-supplied expectation provenance`, severity: "error", source: "property-provenance", - path: contributionPath + path: `state.nodes.${dependencyId}.provenance.reference_expectations` }); + continue; } - if (!hasDeduplication) { + for (const expectationPath of expectationPaths) { + appendExpectationCatalog( + layout, + dependencyId, + path.join(dependencyDir, expectationPath), + expectationIds, + metadata.sha256, + diagnostics + ); + } + } + if (!catalogSupplied) { + // Issue #285: structured catalogs are the sole authority. Most benchmark + // runs intentionally supply none, so record that expectation-backed + // coverage is unavailable while requiring every lens row to omit the field. + diagnostics.push({ + code: "PROPERTY_REFERENCE_EXPECTATION_CATALOG_ABSENT", + message: + "No pinned-reference dependency supplies a structured reference expectation catalog, so lens artifacts must omit reference_expectations", + severity: "warning", + source: "property-provenance", + path: `state.nodes.${node.id}.depends_on` + }); + } + return { ids: expectationIds, catalogSupplied, diagnostics }; +} + +function declaresReferenceExpectationCatalog( + outputs: ReadonlyArray<{ path: string; contract: string }> | undefined, + expectedPath: string +): boolean { + return ( + outputs?.some( + (output) => output.path === expectedPath && output.contract === "ultrafuzz/reference-expectations@2" + ) ?? false + ); +} + +function appendExpectationCatalog( + layout: RunLayout, + dependencyId: string, + catalogPath: string, + expectationIds: Set, + expectedDigest: string, + diagnostics: RuntimeDiagnostic[] +): void { + try { + const stat = fs.lstatSync(catalogPath); + if (!stat.isFile() || stat.isSymbolicLink()) { diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_REQUIRED", - message: `Finding ${JSON.stringify(finding.id)} must declare deduplication.pre_dedup_count`, + code: "PROPERTY_REFERENCE_EXPECTATION_TAMPERED", + message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} is not a regular file`, severity: "error", source: "property-provenance", - path: deduplicationPath + path: catalogPath }); + return; } - - const references = campaignFailureReferences(finding.contributing_backend_failures); - const deduplication = isRecord(finding.deduplication) ? finding.deduplication : undefined; - const preDedupCount = deduplication?.pre_dedup_count; - if (references !== undefined && typeof preDedupCount === "number" && preDedupCount !== references.length) { + const catalogContents = readRegularFileSnapshot(catalogPath, MAX_ARTIFACT_SNAPSHOT_BYTES); + const actualDigest = sha256Bytes(catalogContents); + if (actualDigest !== expectedDigest) { diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_COUNT_MISMATCH", - message: `Finding ${JSON.stringify(finding.id)} reports deduplication.pre_dedup_count ${preDedupCount}, but contributing_backend_failures contains ${references.length} entries`, + code: "PROPERTY_REFERENCE_EXPECTATION_TAMPERED", + message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} does not match its recorded provenance digest`, severity: "error", source: "property-provenance", - path: `${deduplicationPath}.pre_dedup_count` + path: catalogPath }); + return; } - if (references === undefined) continue; - - const resolvedFailures: PartitionedCampaignFailure[] = []; - for (const [referenceIndex, reference] of references.entries()) { - const referencePath = `${contributionPath}[${referenceIndex}]`; - const candidates = - failuresByBackendAndId.get(campaignBackendFailureKey(reference.fuzzer_backend, reference.failure_id)) ?? []; - if (candidates.length === 0) { - diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_REFERENCE_UNKNOWN", - message: `Finding ${JSON.stringify(finding.id)} names unknown contributing backend failure ${JSON.stringify(reference)}`, - severity: "error", - source: "property-provenance", - path: referencePath - }); - continue; - } - if (candidates.length > 1) { - diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_REFERENCE_AMBIGUOUS", - message: `Finding ${JSON.stringify(finding.id)} uses an ambiguous contributing backend failure ${JSON.stringify(reference)}; qualify it with fuzzer_backend and failure_id`, - severity: "error", - source: "property-provenance", - path: referencePath - }); - continue; - } - const failure = candidates[0]!; - if (reference.raw_result_ref !== failure.rawResultRef) { - diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_RAW_RESULT_MISMATCH", - message: `Finding ${JSON.stringify(finding.id)} contribution raw_result_ref must name the authenticated campaign result ${JSON.stringify(failure.rawResultRef)}`, - severity: "error", - source: "property-provenance", - path: `${referencePath}.raw_result_ref` - }); - continue; - } - if (failure.propertyIds.length === 0) { - diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_REFERENCE_UNKNOWN", - message: `Finding ${JSON.stringify(finding.id)} contribution ${JSON.stringify(reference)} does not name a property-derived failure`, - severity: "error", - source: "property-provenance", - path: referencePath - }); - continue; - } - resolvedFailures.push(failure); - - const earlierClaim = claimedBy.get(failure.key); - if (earlierClaim !== undefined) { - diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_DUPLICATE", - message: `Property-derived failure ${JSON.stringify(failure.id)} is claimed more than once across findings`, - severity: "error", - source: "property-provenance", - path: referencePath, - details: { - first_claim: `${findingsPath}#$[${earlierClaim.findingIndex}].contributing_backend_failures[${earlierClaim.referenceIndex}]` - } - }); - } else { - claimedBy.set(failure.key, { findingIndex, referenceIndex }); - } - } - - // Unknown, ambiguous, and non-property references already carry precise - // diagnostics above. Do not derive a partial partition from them. - if (resolvedFailures.length !== references.length) continue; - - if (typeof finding.id !== "string" || !resolvedFailures.some((failure) => failure.id === finding.id)) { + const manifest = readArtifactManifest(layout, dependencyId); + const manifestEntry = manifest.files.find( + (file) => file.path === path.relative(getNodeArtifactDir(layout, dependencyId), catalogPath) + ); + if (manifestEntry?.sha256 !== expectedDigest) { diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_REPRESENTATIVE_MISMATCH", - message: `Finding ${JSON.stringify(finding.id)} must use the ID of one failure in its contributing_backend_failures partition`, + code: "PROPERTY_REFERENCE_EXPECTATION_MANIFEST_MISMATCH", + message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} is not bound to its artifact manifest digest`, severity: "error", source: "property-provenance", - path: `${findingsPath}#$[${findingIndex}].id` + path: path.join(getNodeArtifactDir(layout, dependencyId), "artifact-manifest.json") }); + return; } - - const contributedPropertyIds = [...new Set(resolvedFailures.flatMap((failure) => failure.propertyIds))]; - if (!sameStringSet(propertyIds, contributedPropertyIds)) { + const parsed = validateReferenceExpectationsSchema(parseStrictJsonBytes(catalogContents), catalogPath); + if (!parsed.ok || parsed.value === undefined) { diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_PROPERTY_MISMATCH", - message: `Finding ${JSON.stringify(finding.id)} property_ids must exactly equal the union reported by its contributing_backend_failures`, + code: "PROPERTY_REFERENCE_EXPECTATION_TAMPERED", + message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} failed schema validation`, severity: "error", source: "property-provenance", - path: `${findingsPath}#$[${findingIndex}].property_ids` + path: catalogPath }); + return; } - - const missingBackendCount = resolvedFailures.filter((failure) => failure.fuzzerBackend === undefined).length; - const contributedBackends = [ - ...new Set( - resolvedFailures.flatMap((failure) => (failure.fuzzerBackend === undefined ? [] : [failure.fuzzerBackend])) - ) - ]; - const ownedBackends = findingFuzzerBackendProvenance(finding); - const hasExpectedBackendShape = - contributedBackends.length === 0 - ? !ownedBackends.present - : contributedBackends.length === 1 - ? Object.prototype.hasOwnProperty.call(finding, "fuzzer_backend") - : Object.prototype.hasOwnProperty.call(finding, "fuzzer_backends"); - if ( - !ownedBackends.valid || - missingBackendCount > 0 || - !hasExpectedBackendShape || - !sameStringSet(ownedBackends.backends, contributedBackends) - ) { + const uniqueness = executeSemanticGate("reference-expectation-id-uniqueness", { document: parsed.value }); + if (uniqueness.status === "failed" || uniqueness.status === "requires-context") { diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_BACKEND_MISMATCH", - message: - missingBackendCount > 0 - ? `Finding ${JSON.stringify(finding.id)} cannot bind backend provenance because ${missingBackendCount} contributing failure record(s) omit fuzzer_backend` - : `Finding ${JSON.stringify(finding.id)} fuzzer backend provenance must exactly match its contributing_backend_failures`, + code: "PROPERTY_REFERENCE_EXPECTATION_TAMPERED", + message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} failed semantic validation`, severity: "error", source: "property-provenance", - path: `${findingsPath}#$[${findingIndex}]` + path: catalogPath, + details: + uniqueness.status === "failed" + ? { gate: uniqueness.gate, issues: uniqueness.issues } + : { gate: uniqueness.gate, missing_context: uniqueness.missingContext } }); + return; } - } - - for (const failure of failures) { - if (failure.propertyIds.length === 0 || claimedBy.has(failure.key)) continue; + for (const expectation of parsed.value.expectations) expectationIds.add(expectation.id); + } catch { diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PARTITION_UNCLAIMED", - message: `Property-derived campaign failure ${JSON.stringify(failure.id)} is not claimed by any finding's contributing_backend_failures`, + code: "PROPERTY_REFERENCE_EXPECTATION_MANIFEST_MISMATCH", + message: `Reference expectation catalog for ${JSON.stringify(dependencyId)} has no verifiable artifact manifest`, severity: "error", source: "property-provenance", - path: failure.path + path: catalogPath }); + return; } - return diagnostics; -} - -function campaignFailureReferences(value: unknown): CampaignFailureReference[] | undefined { - if (!Array.isArray(value)) return undefined; - const references: CampaignFailureReference[] = []; - for (const reference of value) { - if ( - isRecord(reference) && - typeof reference.fuzzer_backend === "string" && - typeof reference.failure_id === "string" && - typeof reference.raw_result_ref === "string" - ) { - references.push({ - fuzzer_backend: reference.fuzzer_backend, - failure_id: reference.failure_id, - raw_result_ref: reference.raw_result_ref - }); - } - } - return references; } -function campaignBackendFailureKey(fuzzerBackend: string, failureId: string): string { - return JSON.stringify([fuzzerBackend, failureId]); -} - -function campaignSummaryFailureCountDiagnostics( - artifactDir: string, - summaryPath: string | undefined, - campaigns: readonly PropertyCampaignArtifact[], - findings: readonly Readonly>[], - authenticated?: AuthenticatedArtifactGateSnapshots +/** Validate the explicit priority selection required by the current contract. */ +function verifyImplementationSelectionCoverage( + catalog: PropertiesArtifact, + implementation: ImplementedPropertiesArtifact, + implementationPath: string, + layout: RunLayout ): RuntimeDiagnostic[] { - if (summaryPath === undefined) return []; - const summary = parseCurrentArtifactJson(artifactDir, summaryPath, authenticated); - if (summary === undefined) return []; - if (!isRecord(summary) || !Object.prototype.hasOwnProperty.call(summary, "failure_counts")) return []; - if (!isRecord(summary.failure_counts)) { + const selection = implementation.selection; + if (selection === undefined) { return [ { - code: "CAMPAIGN_SUMMARY_FAILURE_COUNTS_INVALID", - message: - "Declared campaign summary failure_counts must be an object containing pre_deduplication and post_deduplication counts", + code: "PROPERTY_IMPLEMENTATION_SELECTION_MISSING", + message: "Invariant implementation artifacts must declare selection metadata", severity: "error", - source: "campaign-summary", - path: `${summaryPath}#$.failure_counts` + source: "property-provenance", + path: `${implementationPath}#$.selection` } ]; } - const expected = { - pre_deduplication: campaigns.reduce((total, campaign) => total + campaign.failures.length, 0), - post_deduplication: findings.length - } as const; const diagnostics: RuntimeDiagnostic[] = []; - for (const field of ["pre_deduplication", "post_deduplication"] as const) { - const actual = summary.failure_counts[field]; - const population = - field === "pre_deduplication" - ? "total failures across the sibling backend records" - : "objects in the sibling findings.json array"; - if (typeof actual !== "number" || !Number.isSafeInteger(actual) || actual < 0) { - diagnostics.push({ - code: "CAMPAIGN_SUMMARY_FAILURE_COUNT_INVALID", - message: `Declared campaign summary failure_counts.${field} must be a non-negative safe integer equal to the ${population}`, - severity: "error", - source: "campaign-summary", - path: `${summaryPath}#$.failure_counts.${field}` - }); - continue; - } - if (actual !== expected[field]) { - diagnostics.push({ - code: "CAMPAIGN_SUMMARY_FAILURE_COUNT_MISMATCH", - message: `Declared campaign summary failure_counts.${field} reports ${actual}, but the ${population} is ${expected[field]}`, - severity: "error", - source: "campaign-summary", - path: `${summaryPath}#$.failure_counts.${field}` - }); - } + const configuredSelection = readConfiguredInvariantPrioritySelection(layout); + if (configuredSelection === undefined) { + diagnostics.push({ + code: "PROPERTY_IMPLEMENTATION_CONFIG_MISSING", + message: + "Current invariant implementation coverage cannot be verified without resolved invariant priority configuration", + severity: "error", + source: "property-provenance", + path: layout.resolvedConfigPath + }); + } + const priorityOrder = ["high", "medium", "low"] as const; + const thresholdIndex = priorityOrder.indexOf(selection.priority_threshold); + const expectedPriorities = priorityOrder.slice(0, thresholdIndex + 1); + if ( + selection.priorities.length !== expectedPriorities.length || + selection.priorities.some((priority, index) => priority !== expectedPriorities[index]) + ) { + diagnostics.push({ + code: "PROPERTY_IMPLEMENTATION_SELECTION_INVALID", + message: `Implementation selection priorities must include exactly the priorities at or above ${JSON.stringify(selection.priority_threshold)}`, + severity: "error", + source: "property-provenance", + path: `${implementationPath}#$.selection.priorities` + }); + } + if ( + configuredSelection !== undefined && + (selection.priority_threshold !== configuredSelection.priority_threshold || + selection.priorities.length !== configuredSelection.priorities.length || + selection.priorities.some((priority, index) => priority !== configuredSelection.priorities[index])) + ) { + diagnostics.push({ + code: "PROPERTY_IMPLEMENTATION_SELECTION_CONFIG_MISMATCH", + message: "Implementation selection does not match the resolved invariant priority configuration", + severity: "error", + source: "property-provenance", + path: `${implementationPath}#$.selection` + }); } - return diagnostics; -} -function campaignFindingFuzzerBackendDiagnostics( - campaigns: readonly PropertyCampaignArtifact[], - findings: readonly Readonly>[], - findingsPath: string -): RuntimeDiagnostic[] { - const knownBackends = new Set( - campaigns.flatMap((campaign) => (campaign.fuzzer_backend === undefined ? [] : [campaign.fuzzer_backend])) - ); - const inferredByFailureId = new Map>(); - for (const campaign of campaigns) { - if (campaign.fuzzer_backend === undefined) continue; - for (const failure of campaign.failures) { - const backends = inferredByFailureId.get(failure.id) ?? new Set(); - backends.add(campaign.fuzzer_backend); - inferredByFailureId.set(failure.id, backends); - } + const expectedIds = catalog.properties + .filter( + (property) => + selection.priorities.includes(property.priority) || + (configuredSelection?.reference_expectation_selection !== "priority" && + property.reference_expectations !== undefined && + property.reference_expectations.length > 0) + ) + .map((property) => property.id); + const selectedIds = new Set(selection.property_ids); + const expectedIdSet = new Set(expectedIds); + const missingSelectedIds = expectedIds.filter((propertyId) => !selectedIds.has(propertyId)); + const extraSelectedIds = selection.property_ids.filter((propertyId) => !expectedIdSet.has(propertyId)); + const selectionOrderMatches = + selection.property_ids.length === expectedIds.length && + selection.property_ids.every((propertyId, index) => propertyId === expectedIds[index]); + if (missingSelectedIds.length > 0 || extraSelectedIds.length > 0 || !selectionOrderMatches) { + diagnostics.push({ + code: "PROPERTY_IMPLEMENTATION_SELECTION_MISMATCH", + message: `Implementation selection must list every canonical property matching the configured priority and reference-expectation selection policy in catalog order (missing: ${JSON.stringify(missingSelectedIds)}, extra: ${JSON.stringify(extraSelectedIds)})`, + severity: "error", + source: "property-provenance", + path: `${implementationPath}#$.selection.property_ids` + }); } - const diagnostics: RuntimeDiagnostic[] = []; - for (const [findingIndex, finding] of findings.entries()) { - const owned = findingFuzzerBackendProvenance(finding); - if (owned.present && !owned.valid) { - diagnostics.push({ - code: "PROPERTY_FINDING_FUZZER_BACKEND_INVALID", - message: - "Campaign findings must use one non-empty fuzzer_backend string or one non-empty unique fuzzer_backends array, never both", - severity: "error", - source: "property-provenance", - path: `${findingsPath}#$[${findingIndex}]` - }); - continue; - } - if (owned.present) { - const unknownBackends = owned.backends.filter((backend) => !knownBackends.has(backend)); - if (unknownBackends.length > 0) { - diagnostics.push({ - code: "PROPERTY_FINDING_FUZZER_BACKEND_UNKNOWN", - message: `Campaign finding ${JSON.stringify(finding.id)} names backends absent from the sibling result records: ${unknownBackends.map((backend) => JSON.stringify(backend)).join(", ")}`, - severity: "error", - source: "property-provenance", - path: `${findingsPath}#$[${findingIndex}].${Array.isArray(finding.fuzzer_backends) ? "fuzzer_backends" : "fuzzer_backend"}` - }); - } - continue; - } - if ( - typeof finding.id === "string" && - stringArray(finding.property_ids).length > 0 && - (inferredByFailureId.get(finding.id)?.size ?? 0) > 1 - ) { + const recordsById = new Map(implementation.properties.map((record) => [record.property_id, record])); + const recordIds = new Set(recordsById.keys()); + const missingRecords = expectedIds.filter((propertyId) => !recordIds.has(propertyId)); + const extraRecords = implementation.properties + .map((record) => record.property_id) + .filter((propertyId) => !expectedIdSet.has(propertyId)); + if (missingRecords.length > 0 || extraRecords.length > 0) { + diagnostics.push({ + code: "PROPERTY_IMPLEMENTATION_COVERAGE_INCOMPLETE", + message: `Implementation records must cover exactly the selected canonical properties (missing: ${JSON.stringify(missingRecords)}, extra: ${JSON.stringify(extraRecords)})`, + severity: "error", + source: "property-provenance", + path: `${implementationPath}#$.properties` + }); + } + + for (const [recordIndex, record] of implementation.properties.entries()) { + const canonical = catalog.properties.find((property) => property.id === record.property_id); + const expectedReferenceExpectations = canonical?.reference_expectations ?? []; + const actualReferenceExpectations = record.reference_expectations ?? []; + if (!sameStringSet(actualReferenceExpectations, expectedReferenceExpectations)) { diagnostics.push({ - code: "PROPERTY_FINDING_FUZZER_BACKEND_AMBIGUOUS", - message: `Campaign finding ${JSON.stringify(finding.id)} matches failures from several backends and must own an explicit fuzzer_backends array`, + code: "PROPERTY_IMPLEMENTATION_REFERENCE_EXPECTATIONS_MISMATCH", + message: `Implementation record ${JSON.stringify(record.property_id)} must preserve the complete canonical reference expectation ID set`, severity: "error", source: "property-provenance", - path: `${findingsPath}#$[${findingIndex}].id` + path: `${implementationPath}#$.properties[${recordIndex}].reference_expectations` }); } } + + for (const [recordIndex, record] of implementation.properties.entries()) { + if (!expectedIdSet.has(record.property_id) || record.status === "implemented") continue; + if (record.blocker !== undefined) continue; + diagnostics.push({ + code: "PROPERTY_IMPLEMENTATION_BLOCKER_MISSING", + message: `Selected property ${JSON.stringify(record.property_id)} is ${record.status} and must carry an actionable blocker with code, summary, and next_action`, + severity: "error", + source: "property-provenance", + path: `${implementationPath}#$.properties[${recordIndex}].blocker` + }); + } return diagnostics; } +function readConfiguredInvariantPrioritySelection(layout: RunLayout): + | { + priority_threshold: "high" | "medium" | "low"; + priorities: ("high" | "medium" | "low")[]; + reference_expectation_selection?: "priority" | "mandatory"; + } + | undefined { + const invariants = resolvedInvariantConfig(layout); + const priority_threshold = invariants?.propertyPriorityThreshold; + if (priority_threshold === undefined) return undefined; + return { + priority_threshold, + priorities: invariantPropertyPrioritySelection(priority_threshold).priorities, + reference_expectation_selection: invariants?.referenceExpectationSelection + }; +} + function verifyFinalReportPropertyReferences( layout: RunLayout, artifactDir: string, node: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RuntimeDiagnostic[] { const reportOutputs = node.outputs.filter((output) => output.contract === "ultrafuzz/report@3"); @@ -7268,7 +5431,7 @@ function verifyFinalReportCoverageEvidence( report: Record, reportPath: string, markdownPath: string, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RuntimeDiagnostic[] { const diagnostics = unscopedCoverageScoreDiagnosticsInJson(report, reportPath); @@ -7278,8 +5441,10 @@ function verifyFinalReportCoverageEvidence( const producerStatus = plannedContractProducerStatus(layout, node, "ultrafuzz/coverage-evidence@1", attemptAuthority); if (producerStatus === "absent") { - const markdownClaimsCoverage = markdownCoverageScoreOccurrences(markdown).some((occurrence) => - coverageScoreHasContext(occurrence, false) + // A score that names an exact declaration-completeness scope claims + // coverage evidence; unscoped prose scores stay advisory warnings. + const markdownClaimsCoverage = markdownCoverageScoreOccurrences(markdown).some( + (occurrence) => occurrence.scopes.length > 0 ); if (report.coverage_evidence !== undefined || markdownClaimsCoverage) { diagnostics.push({ @@ -7292,7 +5457,6 @@ function verifyFinalReportCoverageEvidence( } return diagnostics; } - if (producerStatus === "unknown" && report.coverage_evidence === undefined) return diagnostics; const evidence = finalizedSingletonAncestorOutput( layout, node, @@ -7480,7 +5644,7 @@ function unscopedCoverageScoreDiagnostics( { code: "UNSCOPED_COVERAGE_SCORE", message: "Published coverage scores must name an exact declaration-completeness scope on the same line", - severity: "error" as const, + severity: "warning" as const, source: "coverage-evidence", path: `${artifactPath}:${occurrence.line}` }, @@ -7538,13 +5702,18 @@ function reportCoverageScoreDiagnostic( path: `${artifactPath}:${occurrence.line}` }; } + // Unscoped prose scores are advisory. This natural-language scan runs only on + // the host, after the in-workflow verifier accepted the attempt, and it + // matches ordinary sentences such as "Recon reached 85% line coverage" or + // "Handlers reachable: 7/9". The typed coverage evidence and scores that name + // an exact scope are checked separately. const percentage = occurrence.kind === "percentage"; return { code: percentage ? "UNSCOPED_COVERAGE_PERCENTAGE" : "UNSCOPED_COVERAGE_FRACTION", message: percentage ? "Coverage percentages must name an exact declaration-completeness scope on the same rendered line" : "Coverage fractions must name an exact declaration-completeness scope on the same rendered line", - severity: "error", + severity: "warning", source: "coverage-evidence", path: `${artifactPath}:${occurrence.line}` }; @@ -8671,7 +6840,7 @@ function verifyFinalReportImplementationCoverage( report: Record, reportPath: string, markdownPath: string, - attemptAuthority?: ArtifactGateAttemptAuthority, + attemptAuthority: ArtifactGateAttemptAuthority, authenticated?: AuthenticatedArtifactGateSnapshots ): RuntimeDiagnostic[] { if ( @@ -9070,7 +7239,7 @@ function reportCampaignSourceFindingIds( function readCampaignFuzzerBackends( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): { value: ReadonlyMap; sourceNodeIds: ReadonlySet; @@ -9193,167 +7362,6 @@ function addReportJoinMismatch( } } -function campaignFindingReferenceDiagnostics( - failures: Array<{ id: string; property_ids?: string[] }>, - findings: Array>, - candidateFindingIds: ReadonlySet, - campaignPath: string, - findingsPath: string -): RuntimeDiagnostic[] { - const findingsById = new Map>(); - for (const [findingIndex, finding] of findings.entries()) { - if (typeof finding.id !== "string") { - continue; - } - const propertyIds = Array.isArray(finding.property_ids) - ? finding.property_ids.filter((propertyId): propertyId is string => typeof propertyId === "string") - : []; - const matches = findingsById.get(finding.id) ?? []; - matches.push({ index: findingIndex, propertyIds }); - findingsById.set(finding.id, matches); - } - // A finding covers a failure when it claims every property that failure - // exercised. Coverage is judged per finding, never against the union of all - // findings: a counterexample that broke two invariants at once is a distinct - // observation, and two single-property findings do not report it. - const findingPropertySets = [...findingsById.entries()].flatMap(([findingId, matches]) => - candidateFindingIds.has(findingId) ? matches.map((match) => new Set(match.propertyIds)) : [] - ); - const isCovered = (failurePropertyIds: readonly string[]): boolean => - findingPropertySets.some((propertySet) => failurePropertyIds.every((propertyId) => propertySet.has(propertyId))); - - const diagnostics: RuntimeDiagnostic[] = []; - for (const [failureIndex, failure] of failures.entries()) { - const failurePropertyIds = failure.property_ids ?? []; - const matchingFindings = findingsById.get(failure.id) ?? []; - if ( - matchingFindings.length > 1 && - (failurePropertyIds.length > 0 || matchingFindings.some((finding) => finding.propertyIds.length > 0)) - ) { - diagnostics.push({ - code: "PROPERTY_FINDING_REFERENCE_AMBIGUOUS", - message: `Property-derived campaign failure ${JSON.stringify(failure.id)} has multiple resulting findings`, - severity: "error", - source: "property-provenance", - path: `${campaignPath}#$.failures[${failureIndex}].id` - }); - continue; - } - const matchingFinding = matchingFindings[0]; - if (failurePropertyIds.length > 0 && matchingFinding === undefined) { - // A campaign legitimately deduplicates many counterexamples of the same - // property into one finding, so a failure need not have a finding sharing - // its ID. What it must have is a finding that claims everything it broke; - // otherwise a violation was observed and then dropped. - if (!isCovered(failurePropertyIds)) { - // Name what is actually wrong. Saying "no finding covers property-1, - // property-2" when property-1 is covered sends the retry after the - // wrong artifact, and the node fails again the same way. - const unclaimed = failurePropertyIds.filter( - (propertyId) => !findingPropertySets.some((propertySet) => propertySet.has(propertyId)) - ); - const quoted = (propertyIds: readonly string[]): string => - propertyIds.map((propertyId) => JSON.stringify(propertyId)).join(", "); - diagnostics.push({ - code: "PROPERTY_FINDING_REFERENCE_MISSING", - message: - unclaimed.length > 0 - ? `Property-derived campaign failure ${JSON.stringify(failure.id)} has no resulting finding covering ${quoted(unclaimed)}` - : `Property-derived campaign failure ${JSON.stringify(failure.id)} broke ${quoted(failurePropertyIds)} together, and no single resulting finding claims that combination`, - severity: "error", - source: "property-provenance", - path: `${campaignPath}#$.failures[${failureIndex}].id` - }); - } - continue; - } - // A deduplicated finding reuses one of its failures' IDs, so it may carry - // more properties than that one failure did. It may never carry fewer: - // dropping a property from the finding that anchors a failure loses the - // violation just as surely as omitting the finding. - const anchorCovers = - matchingFinding !== undefined && - failurePropertyIds.every((propertyId) => matchingFinding.propertyIds.includes(propertyId)); - if (matchingFinding !== undefined && !anchorCovers) { - diagnostics.push({ - code: "PROPERTY_FINDING_REFERENCE_MISMATCH", - message: `Campaign failure ${JSON.stringify(failure.id)} has a resulting finding that drops some of its property_ids`, - severity: "error", - source: "property-provenance", - path: `${findingsPath}#$[${matchingFinding.index}].property_ids` - }); - } - } - - return diagnostics; -} - -/** - * Reports findings that no campaign record explains. This must be judged once - * against the union of every campaign record in the node: when a node runs more - * than one backend, a failure observed by one backend is legitimately absent - * from the other backend's record. - */ -function danglingCampaignFindingDiagnostics( - failureIds: ReadonlySet, - findings: Array>, - findingsPath: string -): RuntimeDiagnostic[] { - const diagnostics: RuntimeDiagnostic[] = []; - for (const [findingIndex, finding] of findings.entries()) { - if (typeof finding.id !== "string" || failureIds.has(finding.id)) { - continue; - } - const propertyIds = Array.isArray(finding.property_ids) - ? finding.property_ids.filter((propertyId): propertyId is string => typeof propertyId === "string") - : []; - if (propertyIds.length === 0) { - continue; - } - diagnostics.push({ - code: "PROPERTY_CAMPAIGN_REFERENCE_MISSING", - message: `Property-derived finding ${JSON.stringify(finding.id)} has no campaign failure with the same ID`, - severity: "error", - source: "property-provenance", - path: `${findingsPath}#$[${findingIndex}].id` - }); - } - return diagnostics; -} - -/** - * Reports findings that attribute a property no counterexample ever reported. - * A deduplicated finding may carry more properties than the single failure whose - * ID it reuses, so the failure-to-finding join cannot judge this; without a - * separate check the campaign could invent a violation the fuzzer never - * observed. Like the dangling check this is judged once against the union of - * every campaign record in the node. - */ -function unobservedFindingPropertyDiagnostics( - observedPropertyIds: ReadonlySet, - findings: Array>, - findingsPath: string -): RuntimeDiagnostic[] { - const diagnostics: RuntimeDiagnostic[] = []; - for (const [findingIndex, finding] of findings.entries()) { - const propertyIds = Array.isArray(finding.property_ids) - ? finding.property_ids.filter((propertyId): propertyId is string => typeof propertyId === "string") - : []; - const unobserved = propertyIds.filter((propertyId) => !observedPropertyIds.has(propertyId)); - if (unobserved.length === 0) { - continue; - } - diagnostics.push({ - code: "PROPERTY_CAMPAIGN_PROPERTY_UNOBSERVED", - message: `Finding ${JSON.stringify(finding.id)} claims ${unobserved.map((propertyId) => JSON.stringify(propertyId)).join(", ")}, which no campaign failure reported`, - severity: "error", - source: "property-provenance", - path: `${findingsPath}#$[${findingIndex}].property_ids` - }); - } - return diagnostics; -} - function sameStringSet(left: readonly string[], right: readonly string[]): boolean { const rightSet = new Set(right); return new Set(left).size === rightSet.size && left.every((value) => rightSet.has(value)); @@ -9362,7 +7370,7 @@ function sameStringSet(left: readonly string[], right: readonly string[]): boole function readCanonicalPropertyCatalog( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): { value?: PropertiesArtifact; path?: string; @@ -9397,7 +7405,7 @@ function readCanonicalPropertyCatalog( function readImplementedProperties( layout: RunLayout, consumer: PlannedGraphNode, - attemptAuthority?: ArtifactGateAttemptAuthority + attemptAuthority: ArtifactGateAttemptAuthority ): { value?: ImplementedPropertiesArtifact; path?: string; diff --git a/packages/runtime/src/controller-source.ts b/packages/runtime/src/controller-source.ts index b696b1c20..d38c47d07 100644 --- a/packages/runtime/src/controller-source.ts +++ b/packages/runtime/src/controller-source.ts @@ -12,7 +12,7 @@ const PROVIDER_SCOPED_SENSITIVE_ENVIRONMENT_DECLARATION = Buffer.from( "utf8" ); const UNTRUSTED_SOURCE = - "controller adapter source must exactly match the packaged stock closure; rerun ultrafuzz init --force"; + "controller adapter source must exactly match the packaged stock closure; rerun ultrafuzz init"; const CONTROLLER_NAMES = "claude codex deepseek environment index kimi opencode openrouter pi provider-home strict-json toml".split(" "); export const STOCK_CONTROLLER_SOURCE_TEMPLATES: Readonly> = Object.freeze( @@ -80,7 +80,7 @@ export function assertProviderScopedSensitiveEnvironmentCapability( if (environment === undefined || !environment.contents.includes(PROVIDER_SCOPED_SENSITIVE_ENVIRONMENT_DECLARATION)) { throw new Error( "sealed controller predates provider-scoped sensitive allowlisted environment handling; " + - "rerun ultrafuzz init --force and start a new run" + "rerun ultrafuzz init and start a new run" ); } } diff --git a/packages/runtime/src/data-governance.ts b/packages/runtime/src/data-governance.ts index 72f6f1d2a..a1438c071 100644 --- a/packages/runtime/src/data-governance.ts +++ b/packages/runtime/src/data-governance.ts @@ -6,6 +6,7 @@ import path from "node:path"; import { parseStrictJsonBytes, readSinglyLinkedRegularFileSnapshotInside } from "@ultrafuzz/artifacts"; import type { ResolvedConfig } from "@ultrafuzz/config"; import { isSensitiveEnvironmentName } from "@ultrafuzz/security"; +import { parse as parseToml } from "smol-toml"; import { retryFallbackProfileIds } from "./retry-chain.js"; import { DATA_DISCLOSURE_ACKNOWLEDGEMENTS_JSON_SCHEMA_ID, @@ -43,6 +44,29 @@ const ROUTE_PROXY_ENV = [ "https_proxy", "no_proxy" ] as const; +// Claude Code reads cloud-provider settings only for the platform a +// CLAUDE_CODE_USE_* flag selects (the flags Claude Code 2.1.284 checks; it +// reads each as set only for 1, true, yes, or on, in any case). Without one an +// ambient AWS_PROFILE or GOOGLE_CLOUD_PROJECT routes nothing, yet pinning it +// made the acknowledged route depend on which shell resumed. +const CLAUDE_PLATFORM_FLAG_SET = /^(?:1|true|yes|on)$/iu; +const CLAUDE_CLOUD_ROUTE_PREFIXES: Readonly> = { + CLAUDE_CODE_USE_ANTHROPIC_AWS: ["AWS_"], + CLAUDE_CODE_USE_ANTHROPIC_GOOGLE_CLOUD: ["CLOUD_ML_", "GOOGLE_"], + CLAUDE_CODE_USE_BEDROCK: ["AWS_"], + CLAUDE_CODE_USE_FOUNDRY: ["AZURE_", "FOUNDRY_"], + CLAUDE_CODE_USE_MANTLE: ["AWS_"], + CLAUDE_CODE_USE_VERTEX: ["CLOUD_ML_", "GOOGLE_"] +}; +const PROVIDER_DESTINATIONS: Readonly> = { + ClaudeAgent: "anthropic", + CodexAgent: "openai", + DeepSeekAgent: "deepseek", + KimiAgent: "moonshot", + OpenCodeAgent: "openrouter", + OpenRouterAgent: "openrouter", + PiAgent: "openrouter" +}; export interface DataGovernanceDestinationPolicy { destination: string; processor: string; @@ -301,102 +325,165 @@ function requiredDestinations(config: ResolvedConfig, graph: PlannedGraph, env: }; } export function modelDestination(agent: string, config: ResolvedConfig, env: NodeJS.ProcessEnv): string { - const builtins: Record = { - CodexAgent: "openai", - ClaudeAgent: "anthropic", - KimiAgent: "moonshot", - DeepSeekAgent: "deepseek", - OpenCodeAgent: "openrouter", - OpenRouterAgent: "openrouter", - PiAgent: "openrouter" - }, - route = effectiveRoute(agent, config, env); - if (route !== undefined) return `model:${agent.toLowerCase().replace("agent", "")}-route-${route}`; - if (builtins[agent] === undefined) throw new Error(`cannot derive a data destination for ${agent}`); - return `model:${builtins[agent]}`; + const routeConfig = providerHomeRouteConfig(agent, config, env); + if ( + routeConfig !== undefined && + config.execution.mode === "cloud" && + routeBearingConfig(agent, routeConfig, env) !== undefined + ) + throw new Error( + "cloud execution cannot use host provider-home routing; select and acknowledge the route through environment variables" + ); + return providerRouteDestination(agent, env, routeConfig); } -function effectiveRoute(agent: string, config: ResolvedConfig, env: NodeJS.ProcessEnv): string | undefined { - const routeEnvironment = effectiveRouteEnvironment(agent, env), - provider = { CodexAgent: "codex", KimiAgent: "kimi", ClaudeAgent: "claude" }[agent]; +/** + * The data destination an agent's model traffic reaches. Plan-time disclosure + * acknowledgement calls this, and the generated Claude, Codex, DeepSeek, Kimi, + * and OpenRouter adapters re-verify each invocation with this same function + * (loaded through the runtime module), so there is no second implementation to + * keep in step. Proxies, cloud-provider variables for an unselected platform, + * and a provider CLI's rewrites of unrelated config sections do not count. + */ +export function providerRouteDestination( + agent: string, + env: Record, + routeConfig?: Uint8Array +): string { + // A CLAUDE_CODE_USE_* flag in Claude's settings env selects the platform for + // the process environment's cloud variables too, and vice versa. + const settingsEnv = + agent === "ClaudeAgent" && routeConfig !== undefined ? claudeSettings(routeConfig)?.env : undefined, + route = routeBearingEnvironment(agent, env, [env, settingsEnv ?? {}]), + config = routeConfig === undefined ? undefined : routeBearingConfig(agent, routeConfig, env); + if (route.length === 0 && config === undefined) { + const destination = PROVIDER_DESTINATIONS[agent]; + if (destination === undefined) throw new Error(`cannot derive a data destination for ${agent}`); + return `model:${destination}`; + } + return `model:${agent.toLowerCase().replace("agent", "")}-route-${sha256Stable({ agent, config: config ?? null, route })}`; +} +/** The provider CLI's own config file, when that file can select this agent's route. */ +function providerHomeRouteConfig(agent: string, config: ResolvedConfig, env: NodeJS.ProcessEnv): Buffer | undefined { + const provider = { CodexAgent: "codex", KimiAgent: "kimi", ClaudeAgent: "claude" }[agent]; if (provider === undefined || (agent === "KimiAgent" && config.agents?.KimiAgent?.auth === "api-key")) - return routeEnvironment.length > 0 ? sha256Stable({ agent, config: null, route: routeEnvironment }) : undefined; + return undefined; const configured = config.agents?.[agent]?.configDir, - selectedRoot = env.ULTRAFUZZ_PROVIDER_HOME_ROOT?.trim(), - userHome = env.HOME?.trim() || os.homedir(), - defaultRoot = path.join( + home = providerHome(agent, provider, configured, env), + routeConfig = path.join(home, agent === "ClaudeAgent" ? "settings.json" : "config.toml"); + if (!fs.existsSync(routeConfig)) return undefined; + return readSinglyLinkedRegularFileSnapshotInside(home, routeConfig, 1024 * 1024, "provider route config"); +} +/** The home directory the stock adapter gives this provider's CLI. */ +function providerHome(agent: string, provider: string, configured: string | undefined, env: NodeJS.ProcessEnv): string { + const selectedRoot = env.ULTRAFUZZ_PROVIDER_HOME_ROOT?.trim(), + userHome = env.HOME?.trim() || os.homedir(); + if (configured) { + const defaultRoot = path.join( env.XDG_STATE_HOME?.trim() || path.join(userHome, ".local", "state"), "ultrafuzz", "provider-homes" - ), - home = configured - ? path.join(selectedRoot || defaultRoot, provider, configured) - : selectedRoot - ? path.join(selectedRoot, provider) - : agent === "CodexAgent" - ? env.CODEX_HOME?.trim() || path.join(userHome, ".codex") - : agent === "KimiAgent" - ? env.KIMI_CODE_HOME?.trim() || env.KIMI_SHARE_DIR?.trim() || path.join(userHome, ".kimi-code") - : env.CLAUDE_CONFIG_DIR?.trim() || path.join(userHome, ".claude"), - routeConfig = path.join(home, agent === "ClaudeAgent" ? "settings.json" : "config.toml"); - let configDigest: string | undefined; - if (fs.existsSync(routeConfig)) { - const bytes = readSinglyLinkedRegularFileSnapshotInside(home, routeConfig, 1024 * 1024, "provider route config"); - const affectsRoute = - agent === "ClaudeAgent" - ? claudeSettingsAffectRoute(bytes) - : agent !== "CodexAgent" || codexConfigAffectsRoute(bytes.toString("utf8")); - if (affectsRoute) configDigest = hash(bytes); - if (configDigest !== undefined && config.execution.mode === "cloud") - throw new Error( - "cloud execution cannot use host provider-home routing; select and acknowledge the route through environment variables" - ); + ); + return path.join(selectedRoot || defaultRoot, provider, configured); } - return routeEnvironment.length > 0 - ? sha256Stable({ agent, config: configDigest ?? null, route: routeEnvironment }) - : configDigest; + if (selectedRoot) return path.join(selectedRoot, provider); + if (agent === "CodexAgent") return env.CODEX_HOME?.trim() || path.join(userHome, ".codex"); + if (agent === "KimiAgent") + return env.KIMI_CODE_HOME?.trim() || env.KIMI_SHARE_DIR?.trim() || path.join(userHome, ".kimi-code"); + return env.CLAUDE_CONFIG_DIR?.trim() || path.join(userHome, ".claude"); } /** - * The Codex CLI rewrites its own config.toml on invocation — marketplace - * `last_updated` timestamps, plugin toggles, and project trust levels — so - * digesting the whole file makes the acknowledged route change the moment the - * CLI first runs in a fresh HOME, which failed every sandbox agent task after - * disclosure (#908). Mirror claudeSettingsAffectRoute: only content that can - * actually redirect traffic — a `model_provider` selection, a - * `[model_providers…]` table, or a `base_url` assignment, the same fields - * codexProviderRouting reads — participates in the route digest. A config - * that gains any of these after acknowledgement still fails closed. + * Environment entries that select where an agent's model traffic goes. A + * Claude cloud prefix counts while any of `selectors` sets its platform flag. */ -function codexConfigAffectsRoute(text: string): boolean { - return ( - /(?:^|\n)\s*(?:model_provider|"model_provider"|'model_provider')\s*=/u.test(text) || - /(?:^|\n)\s*\[[^\]\n]*model_providers[^\]\n]*\]/u.test(text) || - /(?:^|\n)\s*(?:base_url|"base_url"|'base_url')\s*=/u.test(text) +function routeBearingEnvironment( + agent: string, + env: Record, + selectors: ReadonlyArray> +): Array<[string, string]> { + const selected = new Set( + Object.entries(CLAUDE_CLOUD_ROUTE_PREFIXES).flatMap(([flag, prefixes]) => + selectors.some((source) => CLAUDE_PLATFORM_FLAG_SET.test(source[flag]?.trim() ?? "")) ? prefixes : [] + ) + ), + inactivePrefixes = + agent === "ClaudeAgent" + ? Object.values(CLAUDE_CLOUD_ROUTE_PREFIXES) + .flat() + .filter((prefix) => !selected.has(prefix)) + : []; + return effectiveRouteEnvironment(agent, env).filter( + ([name]) => !ROUTE_PROXY_ENV.includes(name as never) && !inactivePrefixes.some((prefix) => name.startsWith(prefix)) ); } -function claudeSettingsAffectRoute(bytes: Buffer): boolean { - const parsed = parseStrictJsonBytes(bytes, { - maxBytes: 1024 * 1024, - maxDepth: 32, - maxItems: 4096, - maxProperties: 4096 - }); - if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) - throw new Error("Claude settings must be a JSON object"); - if (Object.keys(parsed).some((name) => /(?:helper|refresh|credentialexport|processwrapper|proxyauth)$/iu.test(name))) - return true; - const configuredEnv = (parsed as Record).env; - if (configuredEnv === undefined) return false; - if (configuredEnv === null || typeof configuredEnv !== "object" || Array.isArray(configuredEnv)) - throw new Error("Claude settings env must be a JSON object"); - return Object.keys(configuredEnv).some((name) => { - const upper = name.toUpperCase(); - return ( - ["ALL_PROXY", "HTTP_PROXY", "HTTPS_PROXY", "NO_PROXY"].includes(upper) || - (!NON_ROUTING_PROVIDER_ENVIRONMENT_NAMES.has(upper) && - !isCredentialLikeEnvironmentVariableName(upper) && - ROUTE_ENV_PREFIXES.ClaudeAgent!.some((prefix) => upper.startsWith(prefix))) - ); - }); +/** + * The part of a provider CLI's own config file that can redirect traffic, or + * undefined when it selects no route. The CLIs rewrite unrelated sections of + * these files themselves (Codex refreshes marketplace timestamps and project + * trust levels, #908), so only these fields participate: + * - Codex `config.toml`: the selected `model_provider` (a `profile` may select + * it), that provider's `base_url`, `wire_api`, and `env_key`, and the + * top-level `openai_base_url`; + * - Claude `settings.json`: credential/process helper keys and routing `env`; + * - Kimi `config.toml`: the whole file. + * A file these readers cannot parse routes by its exact bytes, as before. + */ +function routeBearingConfig(agent: string, bytes: Uint8Array, env: Record): unknown { + if (agent === "CodexAgent") return codexRouteConfig(bytes); + if (agent === "ClaudeAgent") return claudeRouteConfig(bytes, env); + return hash(bytes); +} +function codexRouteConfig(bytes: Uint8Array): unknown { + let config: Record; + try { + config = parseToml(Buffer.from(bytes).toString("utf8")); + } catch { + return { unparsed: hash(bytes) }; + } + const profile = typeof config.profile === "string" ? record(record(config.profiles)?.[config.profile]) : undefined, + selected = profile?.model_provider ?? config.model_provider; + if (typeof selected !== "string") return undefined; + const provider = record(record(config.model_providers)?.[selected]) ?? {}; + return { + model_provider: selected, + base_url: provider.base_url ?? null, + wire_api: provider.wire_api ?? null, + env_key: provider.env_key ?? null, + // Redirects the built-in openai provider when that is the one selected. + openai_base_url: config.openai_base_url ?? null + }; +} +function claudeRouteConfig(bytes: Uint8Array, env: Record): unknown { + const parsed = claudeSettings(bytes); + if (parsed === undefined) return { unparsed: hash(bytes) }; + const helpers = Object.entries(parsed.settings) + .filter(([name]) => /(?:helper|refresh|credentialexport|processwrapper|proxyauth)$/iu.test(name)) + .sort(([left], [right]) => (left < right ? -1 : left > right ? 1 : 0)), + routeEnv = routeBearingEnvironment("ClaudeAgent", parsed.env, [env, parsed.env]); + return helpers.length === 0 && routeEnv.length === 0 ? undefined : { helpers, env: routeEnv }; +} +/** Claude settings.json, or undefined when it is not an object with an object `env`. */ +function claudeSettings( + bytes: Uint8Array +): { settings: Record; env: Record } | undefined { + let settings: Record | undefined; + try { + settings = record(JSON.parse(Buffer.from(bytes).toString("utf8"))); + } catch { + return undefined; + } + const env = settings === undefined ? undefined : record(settings.env ?? {}); + if (settings === undefined || env === undefined) return undefined; + return { + settings, + env: Object.fromEntries( + Object.entries(env).filter((entry): entry is [string, string] => typeof entry[1] === "string") + ) + }; +} +function record(value: unknown): Record | undefined { + return value !== null && typeof value === "object" && !Array.isArray(value) + ? (value as Record) + : undefined; } export function effectiveRouteEnvironment(agent: string, env: NodeJS.ProcessEnv): Array<[string, string]> { const names = new Set( @@ -465,6 +552,8 @@ export function controllerOwnedGovernancePaths(projectRoot: string, runRoot: str path.join(projectRoot, ".ultrafuzz", "runs"), path.join(projectRoot, ".smithers", "node_modules"), path.join(projectRoot, ".smithers", "workflows"), + // `resume --refresh-controller` renders each refreshed controller here. + path.join(projectRoot, ".smithers", "continuations"), // The workflow engine opens its SQLite database in the target root, so a // launched run leaves engine state in the governed worktree. path.join(projectRoot, "smithers.db"), diff --git a/packages/runtime/src/doctor.ts b/packages/runtime/src/doctor.ts index 0353a8151..e095583cb 100644 --- a/packages/runtime/src/doctor.ts +++ b/packages/runtime/src/doctor.ts @@ -1,9 +1,11 @@ import { execFile } from "node:child_process"; +import fs from "node:fs"; +import os from "node:os"; import path from "node:path"; import { promisify } from "node:util"; import { referencesStatus } from "./references.js"; -import { inspectSmithersInstallation, type SmithersInstallationPosture } from "./smithers.js"; +import { inspectSmithersInstallation } from "./smithers.js"; import { SMITHERS_PACKAGE_NAME, SMITHERS_VERSION } from "./smithers-package.js"; import type { DoctorCheck, @@ -21,6 +23,11 @@ const execFileAsync = promisify(execFile); const REGISTRY_LOOKUP_TIMEOUT_MS = 10_000; +/** Linux `statfs` filesystem type of a tmpfs mount (`TMPFS_MAGIC`). */ +const TMPFS_MAGIC = 0x01021994; + +const MIN_TEMPORARY_DIRECTORY_FREE_BYTES = 2 * 1024 ** 3; + /** Local commands every run needs regardless of which agent backend is selected. */ const REQUIRED_TOOLCHAIN_COMMANDS = ["git", "node", "forge"] as const; @@ -58,15 +65,19 @@ export async function diagnoseProject(input: DoctorInput) { checks.push({ name: "validate", status: validationStatus, - summary: validation.ok - ? "config, topology, prompts, paths, agents, and trust posture pass" - : "configuration validation reported errors; run ultrafuzz validate for detail" + summary: + validationStatus === "ok" + ? "config, topology, prompts, paths, agents, and trust posture pass" + : validationStatus === "warning" + ? "configuration validation passed with warnings; run ultrafuzz validate --json for detail" + : "configuration validation reported errors; run ultrafuzz validate for detail" }); diagnostics.push(...validation.diagnostics); - const openRouterSelected = - resolved.config !== undefined && - activeTopologyAgentRefs(projectRoot, resolved.config, input.topologyPath).includes("OpenRouterAgent"); + // The agents the selected topology can dispatch to, including retry fallbacks. + const selectedAgentRefs = + resolved.config === undefined ? [] : activeTopologyAgentRefs(projectRoot, resolved.config, input.topologyPath); + const openRouterSelected = selectedAgentRefs.includes("OpenRouterAgent"); const openRouterCredential = openRouterSelected ? resolved.config?.agents.OpenRouterAgent?.apiKeyEnv : undefined; const openRouterCredentialReady = !openRouterSelected || (openRouterCredential !== undefined && (env[openRouterCredential] ?? "").trim() !== ""); @@ -96,18 +107,18 @@ export async function diagnoseProject(input: DoctorInput) { }); diagnostics.push(...references.diagnostics); - const agentRefs = configuredAgentRefs(resolved.config?.models.profiles); - const topologyCommands = validation.value?.topology?.required_commands ?? []; - const commandRequirements = [ - ...REQUIRED_TOOLCHAIN_COMMANDS.map((name) => ({ name, required: true })), - ...topologyCommands.map((name) => ({ name, required: true })), - ...agentRefs.map((agentRef) => ({ - name: AGENT_EXECUTABLES[agentRef] ?? agentRef, - // Only demand a CLI for agents whose executable Ultrafuzz actually - // knows; an unrecognised ref is reported without being required. - required: AGENT_EXECUTABLES[agentRef] !== undefined - })) - ].filter((entry, index, entries) => entries.findIndex((candidate) => candidate.name === entry.name) === index); + const requiredByName = new Map(); + for (const name of [...REQUIRED_TOOLCHAIN_COMMANDS, ...(validation.value?.topology?.required_commands ?? [])]) { + requiredByName.set(name, true); + } + for (const agentRef of configuredAgentRefs(resolved.config?.models.profiles)) { + // Every configured agent's CLI is reported, but only a selected agent's + // known CLI is required. Agents can share a CLI, so any requirer wins. + const name = AGENT_EXECUTABLES[agentRef] ?? agentRef; + const required = AGENT_EXECUTABLES[agentRef] !== undefined && selectedAgentRefs.includes(agentRef); + requiredByName.set(name, requiredByName.get(name) === true || required); + } + const commandRequirements = [...requiredByName].map(([name, required]) => ({ name, required })); let probeFailure: string | undefined; const probes = resolved.config === undefined @@ -138,7 +149,7 @@ export async function diagnoseProject(input: DoctorInput) { probeFailure !== undefined ? "required command probe failed in the configured execution environment" : missingTools.length === 0 - ? `${toolchain.length} required commands available in the configured execution environment` + ? `${String(toolchain.filter((entry) => entry.required).length)} required commands available in the configured execution environment` : `missing required commands in the configured execution environment: ${missingTools.join(", ")}` }); if (probeFailure !== undefined) { @@ -157,27 +168,29 @@ export async function diagnoseProject(input: DoctorInput) { }); } - const engineCheck = workflowEngineCheck(installation); - const observedEngineStatus = engineCheck.check.status; - checks.push({ - ...engineCheck.check, - status: "unknown" as const, - summary: - "project-local workflow engine posture is informational and ignored; the pinned operator-owned controller is installed, patched, and sealed at launch" - }); - - const patchCheck = compatibilityPatchCheck(installation); - checks.push({ - ...patchCheck.check, - status: "unknown" as const, - summary: - "project-local compatibility-patch posture is informational and ignored; operator-owned controller patches are sealed at launch" - }); + checks.push( + { + name: "workflow-engine-install", + status: "unknown", + summary: + "project-local workflow engine posture is informational and ignored; the pinned operator-owned controller is installed, patched, and sealed at launch" + }, + { + name: "workflow-engine-patches", + status: "unknown", + summary: + "project-local compatibility-patch posture is informational and ignored; operator-owned controller patches are sealed at launch" + } + ); const latestCheck = registryCheck(latest); checks.push(latestCheck.check); diagnostics.push(...latestCheck.diagnostics); + const temporaryCheck = temporaryDirectoryCheck(os.tmpdir()); + checks.push(temporaryCheck.check); + diagnostics.push(...temporaryCheck.diagnostics); + const value: DoctorValue = { project_root: projectRoot, ok: checks.every((check) => check.status !== "error"), @@ -194,7 +207,10 @@ export async function diagnoseProject(input: DoctorInput) { installed_bin_target: installation.installed_bin_target, bin_path: installation.bin_path, latest_published_version: latest !== undefined && "version" in latest ? latest.version : "unknown", - layout_status: observedEngineStatus, + layout_status: + installation.installed_version === installation.required_version && installation.layout_error === null + ? "ok" + : "error", layout_detail: installation.layout_error, compatibility_patches: installation.compatibility_patches } @@ -202,136 +218,71 @@ export async function diagnoseProject(input: DoctorInput) { return runtimeResult(value.ok, value, diagnostics); } -function workflowEngineCheck(installation: SmithersInstallationPosture): { - check: DoctorCheck; - diagnostics: RuntimeDiagnostic[]; -} { - if (installation.installed_version === null) { - return { - check: { - name: "workflow-engine-install", - status: "error", - summary: `pinned workflow engine ${installation.required_version} is not installed for this project` - }, - diagnostics: [ - { - code: "DOCTOR_WORKFLOW_ENGINE_MISSING", - message: `pinned workflow engine ${installation.required_version} is not installed; it installs automatically on the next run`, - severity: "error", - source: "doctor" - } - ] - }; - } - if (installation.installed_version !== installation.required_version) { - return { - check: { - name: "workflow-engine-install", - status: "error", - summary: `installed workflow engine ${installation.installed_version} does not match the required ${installation.required_version}` - }, - diagnostics: [ - { - code: "DOCTOR_WORKFLOW_ENGINE_VERSION_MISMATCH", - message: `installed workflow engine is ${installation.installed_version} but ${installation.required_version} is required`, - severity: "error", - source: "doctor" - } - ] - }; - } - if (installation.layout_error !== null) { - return { - check: { - name: "workflow-engine-install", - status: "error", - summary: `installed workflow engine layout is not usable: ${installation.layout_error}` - }, - diagnostics: [ - { - code: "DOCTOR_WORKFLOW_ENGINE_LAYOUT_INVALID", - message: `installed workflow engine layout failed validation: ${installation.layout_error}`, - severity: "error", - source: "doctor" - } - ] - }; +/** + * Launch and resume install the workflow engine controller under the OS + * temporary directory, and a native resume keeps its install there for the + * detached engine. Warn when that directory is RAM-backed or nearly full. The + * controller roots are only reported: a live engine may still be using them. + */ +function temporaryDirectoryCheck(directory: string): { check: DoctorCheck; diagnostics: RuntimeDiagnostic[] } { + const warning = (message: string) => ({ + check: { name: "temporary-directory", status: "warning" as const, summary: message }, + diagnostics: [ + { code: "DOCTOR_TEMPORARY_DIRECTORY_CONSTRAINED", message, severity: "warning" as const, source: "doctor" } + ] + }); + let stats: fs.StatsFs; + let roots: string[]; + try { + stats = fs.statfsSync(directory); + roots = fs + .readdirSync(directory, { withFileTypes: true }) + .filter((entry) => entry.isDirectory() && entry.name.startsWith("ultrafuzz-controller-")) + .map((entry) => path.join(directory, entry.name)); + } catch (error) { + return warning( + `could not inspect the temporary directory ${directory}: ${error instanceof Error ? error.message : String(error)}` + ); } - return { - check: { - name: "workflow-engine-install", - status: "ok", - summary: `workflow engine ${installation.installed_version} installed and passing manifest and path validation` - }, - diagnostics: [] - }; + const free = stats.bavail * stats.bsize; + const rootBytes = roots.reduce((total, root) => total + regularFileBytes(root), 0); + const usage = `${formatBytes(free)} free; ${String(roots.length)} ultrafuzz-controller-* ${roots.length === 1 ? "directory holds" : "directories hold"} ${formatBytes(rootBytes)}`; + const problems = [ + ...(stats.type === TMPFS_MAGIC ? ["is a RAM-backed tmpfs"] : []), + ...(free < MIN_TEMPORARY_DIRECTORY_FREE_BYTES ? ["has less than 2 GiB free"] : []) + ]; + return problems.length === 0 + ? { check: { name: "temporary-directory", status: "ok", summary: `${directory}: ${usage}` }, diagnostics: [] } + : warning( + `temporary directory ${directory} ${problems.join(" and ")}; launch and resume install the workflow engine controller there (${usage}). Set TMPDIR to a disk-backed directory with more free space.` + ); } -function compatibilityPatchCheck(installation: SmithersInstallationPosture): { - check: DoctorCheck; - diagnostics: RuntimeDiagnostic[]; -} { - const entries = Object.entries(installation.compatibility_patches); - const incompatible = entries.filter(([, posture]) => posture === "incompatible").map(([name]) => name); - const missing = entries.filter(([, posture]) => posture === "missing").map(([name]) => name); - const unknown = entries.filter(([, posture]) => posture === "unknown").map(([name]) => name); - if (incompatible.length > 0) { - // The next run hard-fails in this state, so doctor must not call it healthy. - return { - check: { - name: "workflow-engine-patches", - status: "error", - summary: `installed engine source is modified or incompatible for: ${incompatible.join(", ")}` - }, - diagnostics: [ - { - code: "DOCTOR_WORKFLOW_ENGINE_PATCHES_INCOMPATIBLE", - message: `installed workflow engine source no longer matches the shape Ultrafuzz patches for: ${incompatible.join(", ")}; reinstall the pinned engine`, - severity: "error", - source: "doctor" - } - ] - }; - } - if (missing.length > 0) { - return { - check: { - name: "workflow-engine-patches", - status: "warning", - summary: `compatibility patches not applied yet: ${missing.join(", ")}; they apply on the next run` - }, - diagnostics: [ - { - code: "DOCTOR_WORKFLOW_ENGINE_PATCHES_PENDING", - message: `required workflow engine compatibility patches are not applied: ${missing.join(", ")}`, - severity: "warning", - source: "doctor" - } - ] - }; +/** Total size of the regular files under a directory, skipping entries that vanish or cannot be read. */ +function regularFileBytes(directory: string): number { + let entries: fs.Dirent[]; + try { + entries = fs.readdirSync(directory, { withFileTypes: true }); + } catch { + return 0; } - if (unknown.length > 0) { - return { - check: { - name: "workflow-engine-patches", - status: "unknown", - summary: `compatibility patch posture unavailable for: ${unknown.join(", ")}` - }, - diagnostics: [] - }; + let total = 0; + for (const entry of entries) { + const entryPath = path.join(directory, entry.name); + if (entry.isDirectory()) total += regularFileBytes(entryPath); + else if (entry.isFile()) { + try { + total += fs.lstatSync(entryPath).size; + } catch { + // Removed while scanning. + } + } } - const upstream = entries.filter(([, posture]) => posture === "upstream").map(([name]) => name); - return { - check: { - name: "workflow-engine-patches", - status: "ok", - summary: - upstream.length === entries.length - ? "the installed engine already provides every patched behavior upstream" - : `compatibility patches applied${upstream.length > 0 ? `; upstream now covers ${upstream.join(", ")}` : ""}` - }, - diagnostics: [] - }; + return total; +} + +function formatBytes(bytes: number): string { + return bytes >= 1024 ** 3 ? `${(bytes / 1024 ** 3).toFixed(1)} GiB` : `${String(Math.round(bytes / 1024 ** 2))} MiB`; } function registryCheck(latest: LatestPublishedEngine | { error: string } | undefined): { diff --git a/packages/runtime/src/dynamic-expansion-retry.ts b/packages/runtime/src/dynamic-expansion-retry.ts index 0f14c86c3..7bb3ce17f 100644 --- a/packages/runtime/src/dynamic-expansion-retry.ts +++ b/packages/runtime/src/dynamic-expansion-retry.ts @@ -61,12 +61,12 @@ export interface DynamicExpansionRetryArchive { * generation, and validate everything that could refuse it. * * Smithers resets the producer and all of its dependents, but the expansion - * manifests live outside Smithers state. Leaving them active makes the next - * workflow render require the canonical source artifact during the gap between - * producer completion and verifier publication. The decision is taken here, - * before the first `timetravel`, so ambiguous or unrecognized manifest state - * fails closed while Smithers state is still untouched. Returns `undefined` - * when no published manifest belongs to a retried source. + * manifests live outside Smithers state. Left in place, they keep the group at + * its published items; withdrawing them lets the group expand again from the + * retried source's new output. The decision is taken here, before the first + * `timetravel`, so ambiguous or unrecognized manifest state fails closed while + * Smithers state is still untouched. Returns `undefined` when no published + * manifest belongs to a retried source. */ export function planDynamicExpansionRetryArchive(input: { projectRoot: string; @@ -113,7 +113,11 @@ export function planDynamicExpansionRetryArchive(input: { ); } const expectedEntries = new Set(manifests.map((manifest) => `${manifest.group_node_id}.json`)); - const unexpectedEntries = fs.readdirSync(manifestDir).filter((entry) => !expectedEntries.has(entry)); + // A dot entry is never a manifest (readExpansionManifests skips it): an interrupted publication's + // temporary file, or the `.expansion.lock` older builds left behind. It moves with the directory. + const unexpectedEntries = fs + .readdirSync(manifestDir) + .filter((entry) => !entry.startsWith(".") && !expectedEntries.has(entry)); if (unexpectedEntries.length > 0) { throw dynamicError( "DYNAMIC_RETRY_EXPANSION_INVALID", diff --git a/packages/runtime/src/dynamic-expansion.ts b/packages/runtime/src/dynamic-expansion.ts index fdb05d9ca..5597b266a 100644 --- a/packages/runtime/src/dynamic-expansion.ts +++ b/packages/runtime/src/dynamic-expansion.ts @@ -227,12 +227,8 @@ export function loadOrCreateDynamicExpansion(input: { const runRoot = path.resolve(input.runRoot); assertPathInside(runRoot, input.sourceArtifactPath, "dynamic source artifact"); assertPathInside(runRoot, input.templatePath, "dynamic prompt template"); - assertNoSymlinkComponents(runRoot, input.sourceArtifactPath, "dynamic source artifact"); assertNoSymlinkComponents(runRoot, input.templatePath, "dynamic prompt template"); - assertRegularFileInside(runRoot, input.sourceArtifactPath, "dynamic source artifact"); assertRegularFileInside(runRoot, input.templatePath, "dynamic prompt template"); - const sourceBytes = fs.readFileSync(input.sourceArtifactPath); - const sourceDigest = sha256Bytes(sourceBytes); const actualTemplateDigest = sha256Bytes(fs.readFileSync(input.templatePath)); if (actualTemplateDigest !== input.templateDigest) { throw dynamicError("DYNAMIC_TEMPLATE_CHANGED", `Dynamic group ${input.groupNodeId} prompt template changed`, { @@ -248,88 +244,99 @@ export function loadOrCreateDynamicExpansion(input: { assertNoSymlinkComponents(runRoot, manifestDir, "dynamic expansion manifest directory"); const manifestPath = path.join(manifestDir, `${validateSafeId(input.groupNodeId, "dynamic group node ID")}.json`); const sourceArtifactRelativePath = path.relative(runRoot, input.sourceArtifactPath).split(path.sep).join("/"); - return withDynamicExpansionLock(manifestDir, () => { - const priorManifests = readExpansionManifests(manifestDir); - assertManifestSetMatchesInput(priorManifests, { - runId: input.runId, - maxDynamicNodes: input.maxDynamicNodes, - reservedNodeIds: input.reservedNodeIds - }); - const existing = priorManifests.find((manifest) => manifest.group_node_id === input.groupNodeId); - if (existing !== undefined) { - assertCompatibleManifest(existing, { - runId: input.runId, - groupNodeId: input.groupNodeId, - sourceNodeId: input.sourceNodeId, - sourceAttemptId: input.sourceAttemptId, - sourceArtifactPath: sourceArtifactRelativePath, - sourceDigest, - templateDigest: input.templateDigest, - templateFingerprint: input.templateFingerprint, - sourcePath: input.sourcePath, - keyPath: input.keyPath, - nodeIdTemplate: input.nodeIdTemplate, - maxDynamicNodes: input.maxDynamicNodes - }); - return existing; - } - - let sourceDocument: unknown; - try { - sourceDocument = JSON.parse(sourceBytes.toString("utf8")) as unknown; - } catch (error) { - throw dynamicError("DYNAMIC_SOURCE_JSON_INVALID", `Dynamic source artifact is not valid JSON`, { - groupNodeId: input.groupNodeId, - reason: error instanceof Error ? error.message : String(error) - }); - } - const reserved = new Set(input.reservedNodeIds ?? []); - for (const manifest of priorManifests) { - for (const item of manifest.items) reserved.add(item.node_id); - } - const manifest = planDynamicExpansion({ + // No lock (#1142): the one this replaced was never reclaimed, so a process killed while holding it + // failed every later render and lifecycle admission. Published manifests are never rewritten, so + // reads need none; on a filesystem without hard links a concurrent reader can still catch one + // mid-publication and fail that read. Creation assumes one renderer per run (Smithers refuses to + // resume a run whose driver is live before it renders). Renderers that load the same outputs + // publish identical bytes, which publishFileDurableExclusive accepts. Two that create different + // groups at once can take the same `sequence`; the re-read after publication detects that but + // cannot undo it, and the run then needs manual repair. + const priorManifests = readExpansionManifests(manifestDir); + assertManifestSetMatchesInput(priorManifests, { + runId: input.runId, + maxDynamicNodes: input.maxDynamicNodes, + reservedNodeIds: input.reservedNodeIds + }); + const existing = priorManifests.find((manifest) => manifest.group_node_id === input.groupNodeId); + if (existing !== undefined) { + // A published manifest is the group's membership. `source.output_sha256` records the bytes it + // was planned from; the live source is not consulted again, because a reset re-runs the source, + // whose agent attempt wipes that artifact and then writes a new one. + assertCompatibleManifest(existing, { runId: input.runId, groupNodeId: input.groupNodeId, sourceNodeId: input.sourceNodeId, sourceAttemptId: input.sourceAttemptId, sourceArtifactPath: sourceArtifactRelativePath, - sourceDigest, - sourceDocument, + templateDigest: input.templateDigest, + templateFingerprint: input.templateFingerprint, sourcePath: input.sourcePath, keyPath: input.keyPath, nodeIdTemplate: input.nodeIdTemplate, - templateDigest: input.templateDigest, - templateFingerprint: input.templateFingerprint, - maxDynamicNodes: input.maxDynamicNodes, - sequence: priorManifests.length, - alreadyExpandedNodes: priorManifests.reduce((sum, entry) => sum + entry.items.length, 0), - reservedNodeIds: reserved - }); - // Validate the complete candidate set in memory first: a manifest that would make the set - // invalid must never reach durable storage, otherwise every automatic resume keeps failing - // until an operator removes or repairs the published file by hand. - const candidateSet = [...priorManifests, manifest]; - validateManifestSet(candidateSet, manifestDir); - assertManifestSetMatchesInput(candidateSet, { - runId: input.runId, - maxDynamicNodes: input.maxDynamicNodes, - reservedNodeIds: input.reservedNodeIds + maxDynamicNodes: input.maxDynamicNodes }); - publishFileDurableExclusive(manifestDir, `${input.groupNodeId}.json`, `${JSON.stringify(manifest, null, 2)}\n`); - const publishedManifests = readExpansionManifests(manifestDir); - assertManifestSetMatchesInput(publishedManifests, { - runId: input.runId, - maxDynamicNodes: input.maxDynamicNodes, - reservedNodeIds: input.reservedNodeIds + return existing; + } + + assertNoSymlinkComponents(runRoot, input.sourceArtifactPath, "dynamic source artifact"); + assertRegularFileInside(runRoot, input.sourceArtifactPath, "dynamic source artifact"); + const sourceBytes = fs.readFileSync(input.sourceArtifactPath); + let sourceDocument: unknown; + try { + sourceDocument = JSON.parse(sourceBytes.toString("utf8")) as unknown; + } catch (error) { + throw dynamicError("DYNAMIC_SOURCE_JSON_INVALID", `Dynamic source artifact is not valid JSON`, { + groupNodeId: input.groupNodeId, + reason: error instanceof Error ? error.message : String(error) }); - const published = publishedManifests.find((candidate) => candidate.group_node_id === input.groupNodeId); - if (published === undefined || !fs.existsSync(manifestPath)) { - throw dynamicError("DYNAMIC_MANIFEST_PUBLISH_FAILED", "Dynamic expansion manifest publication failed", { - groupNodeId: input.groupNodeId - }); - } - return published; + } + const reserved = new Set(input.reservedNodeIds ?? []); + for (const manifest of priorManifests) { + for (const item of manifest.items) reserved.add(item.node_id); + } + const manifest = planDynamicExpansion({ + runId: input.runId, + groupNodeId: input.groupNodeId, + sourceNodeId: input.sourceNodeId, + sourceAttemptId: input.sourceAttemptId, + sourceArtifactPath: sourceArtifactRelativePath, + sourceDigest: sha256Bytes(sourceBytes), + sourceDocument, + sourcePath: input.sourcePath, + keyPath: input.keyPath, + nodeIdTemplate: input.nodeIdTemplate, + templateDigest: input.templateDigest, + templateFingerprint: input.templateFingerprint, + maxDynamicNodes: input.maxDynamicNodes, + sequence: priorManifests.length, + alreadyExpandedNodes: priorManifests.reduce((sum, entry) => sum + entry.items.length, 0), + reservedNodeIds: reserved }); + // Validate the complete candidate set in memory first: a manifest that would make the set + // invalid must never reach durable storage, otherwise every automatic resume keeps failing + // until an operator removes or repairs the published file by hand. + const candidateSet = [...priorManifests, manifest]; + validateManifestSet(candidateSet, manifestDir); + assertManifestSetMatchesInput(candidateSet, { + runId: input.runId, + maxDynamicNodes: input.maxDynamicNodes, + reservedNodeIds: input.reservedNodeIds + }); + publishFileDurableExclusive(manifestDir, `${input.groupNodeId}.json`, `${JSON.stringify(manifest, null, 2)}\n`); + const publishedManifests = readExpansionManifests(manifestDir); + assertManifestSetMatchesInput(publishedManifests, { + runId: input.runId, + maxDynamicNodes: input.maxDynamicNodes, + reservedNodeIds: input.reservedNodeIds + }); + const published = publishedManifests.find((candidate) => candidate.group_node_id === input.groupNodeId); + if (published === undefined || !fs.existsSync(manifestPath)) { + throw dynamicError("DYNAMIC_MANIFEST_PUBLISH_FAILED", "Dynamic expansion manifest publication failed", { + groupNodeId: input.groupNodeId + }); + } + return published; } export function dynamicStorageId(groupNodeId: string, generatedNodeId: string): string { @@ -778,61 +785,6 @@ function assertManifestSetMatchesInput( } } -function withDynamicExpansionLock(manifestDir: string, operation: () => T): T { - const lockPath = path.join(manifestDir, ".expansion.lock"); - const deadline = Date.now() + 5_000; - const token = `${process.pid}:${crypto.randomBytes(16).toString("hex")}`; - let descriptor: number | undefined; - while (descriptor === undefined) { - try { - descriptor = fs.openSync(lockPath, "wx", 0o600); - fs.writeFileSync(descriptor, `${token}\n`, "utf8"); - } catch (error) { - if (!isAlreadyExistsError(error)) throw error; - let stat: fs.Stats; - try { - stat = fs.lstatSync(lockPath); - } catch (statError) { - // The holder may release between our exclusive-create failure and the - // inspection. That is ordinary lock contention, not a run failure. - if (isNoEntryError(statError)) continue; - throw statError; - } - if (stat.isSymbolicLink() || !stat.isFile()) { - throw dynamicError("DYNAMIC_EXPANSION_LOCK_INVALID", "Dynamic expansion lock is unsafe", { lockPath }); - } - // Never steal a lock based on age. Between an age check and unlink, the - // observed inode can disappear and a new owner can publish a fresh lock - // at the same path. Only the token-owning holder releases the lock; - // contenders fail after the bounded wait and leave recovery explicit. - if (Date.now() >= deadline) { - throw dynamicError("DYNAMIC_EXPANSION_LOCKED", "Dynamic expansion is already being materialized", { - lockPath - }); - } - Atomics.wait(new Int32Array(new SharedArrayBuffer(4)), 0, 0, 10); - } - } - try { - return operation(); - } finally { - fs.closeSync(descriptor); - try { - if (fs.readFileSync(lockPath, "utf8").trim() === token) fs.unlinkSync(lockPath); - } catch { - // A missing lock after the operation cannot weaken manifest validation. - } - } -} - -function isAlreadyExistsError(error: unknown): boolean { - return error instanceof Error && "code" in error && error.code === "EEXIST"; -} - -function isNoEntryError(error: unknown): boolean { - return error instanceof Error && "code" in error && error.code === "ENOENT"; -} - type ManifestFailure = (reason: string, details?: Record) => never; function assertExactKeys( @@ -933,7 +885,6 @@ function assertCompatibleManifest( sourceNodeId: string; sourceAttemptId: string; sourceArtifactPath: string; - sourceDigest: string; templateDigest: string; templateFingerprint: string; sourcePath: string; @@ -948,7 +899,6 @@ function assertCompatibleManifest( ["source node ID", manifest.source.node_id, expected.sourceNodeId], ["source attempt ID", manifest.source.attempt_id, expected.sourceAttemptId], ["source artifact", manifest.source.artifact_path, expected.sourceArtifactPath], - ["source output", manifest.source.output_sha256, expected.sourceDigest], ["source path", manifest.source.json_path, expected.sourcePath], ["key path", manifest.template.key_path, expected.keyPath], ["node ID template", manifest.template.node_id, expected.nodeIdTemplate], diff --git a/packages/runtime/src/dynamic-runtime.ts b/packages/runtime/src/dynamic-runtime.ts index c71cbc8ae..0cd1b04bd 100644 --- a/packages/runtime/src/dynamic-runtime.ts +++ b/packages/runtime/src/dynamic-runtime.ts @@ -12,8 +12,7 @@ import { sha256Bytes, validateNodeReference, validateSafeId, - writeFileDurable, - writeJsonDurable + writeFileDurable } from "@ultrafuzz/artifacts"; import { renderPrompt, type PromptConcreteNode, type PromptGraphNode } from "@ultrafuzz/prompts"; @@ -183,9 +182,11 @@ function deriveDynamicRuntime( if (mode === "publish") { // Publish tasks first. A concurrent synchronizer may temporarily skip an // unknown graph node, while the reverse ordering could finalize a graph node - // against stale task identity. Both files are themselves atomically replaced. - writeJsonDurable(tasksPath, runtimeTaskDocument); - writeJsonDurable(graphPath, runtimeGraph); + // against stale task identity. Both files are themselves atomically replaced, + // and only when their bytes change: this runs on every render, and most + // renders re-derive exactly the documents already on disk. + writeJsonDurableIfChanged(tasksPath, runtimeTaskDocument); + writeJsonDurableIfChanged(graphPath, runtimeGraph); } else { const observedTasks = readRecord(tasksPath); const observedGraph = readPlannedGraph(graphPath); @@ -330,6 +331,16 @@ function instantiateDynamicGraphNode(input: { }; } +/** + * Continuation lets independent tasks settle; it does not make a strategy's required inputs + * optional. Only the review group reconciles partial results, so only its tasks treat a continuing + * producer's output as optional (#1120). The compiler and dynamic lowering share this one rule; the + * task-manifest gate checks only that optional inputs come from continuing producers. + */ +export function reconcilesPartialResults(task: Pick): boolean { + return task.metadata.node.group === "review"; +} + function lowerTaskDynamicDependencies( task: CompiledSmithersTask, groups: readonly CompiledSmithersDynamicGroup[], @@ -370,7 +381,7 @@ function lowerTaskDynamicDependencies( dependencies.push(...generated.map((candidate) => candidate.attemptId)); dependencySmithersNodeIds.push(...generated.map((candidate) => candidate.verifierSmithersNodeId)); dependencyArtifactDirs.push(...generated.map((candidate) => candidate.artifactDir)); - if (group.continueOnFail && task.metadata.node.group === "review") { + if (group.continueOnFail && reconcilesPartialResults(task)) { optionalDependencyArtifactDirs.push(...generated.map((candidate) => candidate.artifactDir)); } concreteNodeIds.push(...manifest.items.map((item) => item.node_id)); @@ -638,6 +649,12 @@ function readPlannedGraph(graphPath: string): PlannedGraph { return value as PlannedGraph; } +/** Same bytes as `writeJsonDurable`, skipping the replace and fsyncs when the file already holds them. */ +function writeJsonDurableIfChanged(filePath: string, value: unknown): void { + const bytes = `${JSON.stringify(value, null, 2)}\n`; + if (fs.readFileSync(filePath, "utf8") !== bytes) writeFileDurable(filePath, bytes); +} + function readRecord(filePath: string): Record { const value = JSON.parse(fs.readFileSync(filePath, "utf8")) as unknown; if (typeof value !== "object" || value === null || Array.isArray(value)) { diff --git a/packages/runtime/src/final-report-markdown.ts b/packages/runtime/src/final-report-markdown.ts index e60e04716..3a8909aed 100644 --- a/packages/runtime/src/final-report-markdown.ts +++ b/packages/runtime/src/final-report-markdown.ts @@ -21,7 +21,6 @@ import { redactSecretsInText, type SecretScanMode } from "@ultrafuzz/security"; export const MAX_FINAL_REPORT_JSON_BYTES = 64 * 1024 * 1024; export const MAX_FINAL_REPORT_MARKDOWN_BYTES = 16 * 1024 * 1024; -const SAFE_REPORT_RELATIVE_LINK_PATTERN = /^\.\.\/(?:(?!\.\.?\/)[A-Za-z0-9._-]+\/)+(?!\.\.?$)[A-Za-z0-9._-]+$/u; /** * Secret placeholder for the public projection only. Two constraints pick it: * @@ -29,10 +28,9 @@ const SAFE_REPORT_RELATIVE_LINK_PATTERN = /^\.\.\/(?:(?!\.\.?\/)[A-Za-z0-9._-]+\ * the placeholder must be a fixed point of the redaction pass. The key-name assignment rule's * unquoted value class stops at whitespace, `,`, `;`, `]`, and `}`, so a placeholder containing * any of those is re-redacted on the next pass (`token=[redacted]` becomes `token=[redacted]]`). - * - The final-review Markdown gate rejects raw HTML, images, and links outside fenced code, and - * redacted values land unescaped in inline code (run summary values, coverage paths, source - * nodes), so the placeholder must not read as HTML (``), a link (`[redacted](`), - * an image, or emphasis (`*`, `_`). + * - The final-review Markdown gate rejects raw HTML outside fenced code, including inside inline + * code, where redacted values land unescaped (run summary values, coverage paths, source nodes), + * so the placeholder must not read as HTML (``). * * A bare uppercase word satisfies both. Every other redaction keeps the security package's default. */ @@ -260,9 +258,11 @@ function finalReportMarkdownDirectiveViolation(markdown: string, report: JsonRec if (!markdown.includes("\n## Property provenance\n")) { return "missing property provenance"; } + // Report prose is preserved byte-for-byte from upstream artifacts that the agent cannot repair, so + // the only rule left is one escaped prose cannot match: publicProse escapes `<`, and only + // unescaped inline-code values can still carry raw HTML. const prose = markdownOutsideFencedCode(markdown).replace(//giu, ""); - const proseViolation = finalReportProseDirectiveViolation(prose); - if (proseViolation !== undefined) return proseViolation; + if (/<[A-Za-z][^>]*>/u.test(prose)) return "contains raw HTML outside fenced code"; const rendered = renderedIssues(Array.isArray(report.issues) ? report.issues.filter(isRecord) : []); const expectedHeadings = rendered.map(renderedIssueHeading); const headings = markdown.split("\n").filter((line) => line.startsWith("## [")); @@ -347,31 +347,6 @@ function completionFindingsViolation( return undefined; } -function finalReportProseDirectiveViolation(prose: string): string | undefined { - // Critical is not a supported report severity, but the word remains valid in explanatory prose - // (for example, "a critical invariant"). Reject only a standalone severity-like label rather than - // rewriting or discarding the validated finding text. - const forbiddenPatterns: ReadonlyArray = [ - [/(?:^|\n)(?:#{1,6}\s+|-\s+)?(?:\*\*)?Critical(?:\*\*)?\s*$/imu, "contains the unsupported Critical severity"], - [/(?:^|\n)#### Sources\s*$/imu, "contains a legacy Sources section"], - [/\*\*Source (?:Node|Property) Id\*\*/iu, "contains a legacy source identifier field"], - [/(?:^|\n)- \*\*Item \d+\*\*/imu, "contains a legacy numbered-item field"], - [/(?:^|\n)## (?:Executive summary|Issue index|Additional report data)\s*$/imu, "contains a legacy report section"], - [/(?:^|\n)#{3,6} (?:Lifecycle|Strategy|Strategy provenance)\s*$/imu, "contains a legacy issue subsection"], - [ - /(?:^|\n)- (?:Strategy loops|Audit profile catalog digest|Topology digest|Prompt digest|Expanded graph fingerprint):/imu, - "contains legacy run metadata" - ], - [/<[A-Za-z][^>]*>/u, "contains raw HTML outside fenced code"], - [/!\[[^\]]*\]\(/u, "contains an embedded image outside fenced code"], - [ - /(? pattern.test(prose))?.[1]; -} - function validateReport(report: unknown): JsonRecord { const serialized = `${JSON.stringify(report)}\n`; const validation = validateArtifactContract("ultrafuzz/report@3", serialized, "report.json"); @@ -695,7 +670,6 @@ function renderCanonicalReport(report: JsonRecord, goalSearchCoverage: unknown): lines, isRecord(report.run_metadata) ? report.run_metadata.artifact_validation_warnings : undefined ); - appendAuditContext(lines, report.audit_context); const campaignDidNotRun = appendCampaignOutcome(lines, report.campaign_outcome); appendCoverageEvidence(lines, report.coverage_evidence); const goalCoverage = summarizeGoalSearchCoverage(goalSearchCoverage); @@ -838,8 +812,10 @@ function appendArtifactValidationWarnings(lines: string[], value: unknown): void "" ); for (const warning of value.filter(isRecord)) { + // Codes render as plain text. Only `](` is escaped, so the bytes of real gate codes do not change. + const code = inlineValue(warning.code).replaceAll("](", "]\\("); lines.push( - `- ${inlineValue(warning.code)} — \`${inlineValue(warning.artifact_path)}#${inlineValue(warning.field_path)}\`: ${publicProse(String(warning.message))}` + `- ${code} — \`${inlineValue(warning.artifact_path)}#${inlineValue(warning.field_path)}\`: ${publicProse(String(warning.message))}` ); if (warning.source_path !== undefined) lines.push(` - Available context: \`${inlineValue(warning.source_path)}\``); } @@ -961,29 +937,6 @@ function appendRunSummary(lines: string[], metadata: JsonRecord): void { } } -function appendAuditContext(lines: string[], value: unknown): void { - if (!isRecord(value)) return; - const threat = recordField(value, "threat_model"); - const goalPlan = recordField(value, "goal_plan"); - const threatMarkdown = safeReportLink(threat?.markdown); - const threatJson = safeReportLink(threat?.json); - const goalPlanJson = safeReportLink(goalPlan?.json); - if (threatMarkdown === undefined && threatJson === undefined && goalPlanJson === undefined) return; - lines.push("", "## Audit context", ""); - if (threatMarkdown !== undefined || threatJson !== undefined) { - const links = [ - threatMarkdown === undefined ? undefined : `[THREAT_MODEL.md](${threatMarkdown})`, - threatJson === undefined ? undefined : `[threat-model.json](${threatJson})` - ].filter((entry): entry is string => entry !== undefined); - lines.push(`- Threat model: ${links.join("; ")}`); - } - if (goalPlanJson !== undefined) lines.push(`- Goal plan: [goal-plan.json](${goalPlanJson})`); -} - -function safeReportLink(value: unknown): string | undefined { - return typeof value === "string" && SAFE_REPORT_RELATIVE_LINK_PATTERN.test(value) ? value : undefined; -} - function appendProductionIssue(lines: string[], rendered: RenderedIssue): void { const { issue } = rendered; lines.push("", renderedIssueHeading(rendered), "", publicProse(issueDescription(issue)), "", "### Severity", ""); @@ -1522,6 +1475,7 @@ function recordTitle(record: JsonRecord, fallback: string): string { return typeof record.title === "string" && record.title.trim().length > 0 ? record.title.trim() : fallback; } +/** Escaping `(` after every `]` keeps byte-preserved prose from forming an inline link or image. */ function publicProse(value: string): string { return value .replace(/\s+/gu, " ") @@ -1533,6 +1487,7 @@ function publicProse(value: string): string { .replaceAll("!", "\\!") .replaceAll("#", "\\#") .replaceAll("~", "\\~") + .replaceAll("](", "]\\(") .replaceAll("<", "<") .replaceAll(">", ">"); } diff --git a/packages/runtime/src/index.ts b/packages/runtime/src/index.ts index 856193a23..cb8766908 100644 --- a/packages/runtime/src/index.ts +++ b/packages/runtime/src/index.ts @@ -29,7 +29,6 @@ export * from "./runtime-semantic-gates.js"; export * from "./schema-registry.js"; export * from "./semantic-artifact-context.js"; export * from "./semantic-gates.js"; -export * from "./severity-matrix.js"; export * from "./smithers.js"; export * from "./smithers-package.js"; export * from "./smithers-attempt-authority.js"; diff --git a/packages/runtime/src/init.ts b/packages/runtime/src/init.ts index 425960466..708911bb6 100644 --- a/packages/runtime/src/init.ts +++ b/packages/runtime/src/init.ts @@ -1,4 +1,3 @@ -import crypto from "node:crypto"; import fs from "node:fs"; import path from "node:path"; @@ -16,74 +15,14 @@ import { } from "@ultrafuzz/config"; import { builtInPromptRelativePaths, scaffoldPrompts } from "@ultrafuzz/prompts"; import { defaultReferenceCatalogYaml } from "@ultrafuzz/references"; -import { AGENT_REGISTRY_RELATIVE_PATH, agentRegistryRegisters, inspectAgentRegistry } from "./agent-registry.js"; +import { STOCK_CONTROLLER_SOURCE_TEMPLATES } from "./controller-source.js"; import { loadRuntimeTemplate } from "./runtime-template.js"; import { migrateStockSmithers032PackageManifest, renderSmithersPackageJson } from "./smithers-package.js"; -import type { InitProjectInput, InitProjectResult, RuntimeDiagnostic } from "./types.js"; +import type { InitProjectInput, InitProjectResult } from "./types.js"; import { configDiagnostics, runtimeFailure, runtimeResult, toProjectRelative } from "./utils.js"; const DEFAULT_TOPOLOGY = fs.readFileSync(packagedTopology("default").path, "utf8"); -const MAX_AGENT_ADAPTER_REVIEW_BYTES = 256 * 1024; -const AGENT_TEMPLATES = [ - { - file: "claude.ts", - template: "smithers/agents/claude.tsx", - ref: "ClaudeAgent", - stock032Sha256: ["f2b97c9b57aa45bdc3b42be20d7a3baddc0086a2bae95c161b841599f3262232"] - }, - { - file: "codex.ts", - template: "smithers/agents/codex.tsx", - ref: "CodexAgent", - stock032Sha256: [ - "b932fb7da3c05fdc662f60359e8a751aaabd236ca4072dfeaade1a7bb25a01b5", - "e7e845b2bccf5b7d41a3f0cedfaec7a1457580126513ff96f282a36307c7da98", - "7865f1be1715d36d016c7b2814081b70e70a9aca7e30d5b41f5d91bf2337f681" - ] - }, - { - file: "deepseek.ts", - template: "smithers/agents/deepseek.tsx", - ref: "DeepSeekAgent", - stockSha256: new Set([ - "c23a03c84e2f62d2e6b23ee7b27b1464a633fe20bb5b91c34d2c93d37dcf7e35", - "65bf43f333cbced8ff0157e942d3c78267d8245d6c0041e053c8577a463e7407" - ]) - }, - { - file: "kimi.ts", - template: "smithers/agents/kimi.tsx", - ref: "KimiAgent", - stockSha256: new Set([ - "fdbeaad6ea55122da50e9c9d86ac58a8b6377f419f8e78df23fb1fb401924e0b", - "6de5f4b00b54b533f8fc1aa0628f5a402dbb6521a4584852d83786ee59dcb2fd", - "25c499f8631db6e2529b046b5d2243119b456696c7a4a729345baa6f21f4c4f5", - "f790a3f121da84049032cfc5bf5d300f7bd56e9e3b15d0a51df5ea1b275de75f" - ]) - }, - { - file: "opencode.ts", - template: "smithers/agents/opencode.tsx", - ref: "OpenCodeAgent", - stockSha256: new Set() - }, - { - file: "openrouter.ts", - template: "smithers/agents/openrouter.tsx", - ref: "OpenRouterAgent", - stock032Sha256: [] - }, - { - file: "pi.ts", - template: "smithers/agents/pi.tsx", - ref: "PiAgent", - // No stock digest yet: this adapter has never shipped in a released - // scaffold, so any pi.ts already on disk is the operator's and is preserved. - stockSha256: new Set() - } -] as const; - /** * Canonical artifact JSON Schemas scaffolded into the project. * @@ -154,10 +93,6 @@ export function initProject(input: InitProjectInput) { try { const stockSmithersPackageMigration = input.force === true ? undefined : prepareStockSmithers032PackageMigration(projectRoot); - const upgradedStockAdapters = - stockSmithersPackageMigration === undefined - ? new Set() - : upgradeStockSmithers032Adapters(projectRoot, created, preserved, overwritten); writeProjectFile( projectRoot, "ultrafuzz.toml", @@ -205,63 +140,19 @@ export function initProject(input: InitProjectInput) { preserved, overwritten ); - writeProjectFile( - projectRoot, - ".smithers/agents/index.ts", - loadRuntimeTemplate("smithers/agents/index.tsx"), - input.force === true, - created, - preserved, - overwritten - ); - writeProjectFile( - projectRoot, - ".smithers/agents/toml.ts", - loadRuntimeTemplate("smithers/agents/toml.tsx"), - input.force === true, - created, - preserved, - overwritten - ); - writeProjectFile( - projectRoot, - ".smithers/agents/environment.ts", - loadRuntimeTemplate("smithers/agents/environment.tsx"), - input.force === true, - created, - preserved, - overwritten - ); - writeProjectFile( - projectRoot, - ".smithers/agents/provider-home.ts", - loadRuntimeTemplate("smithers/agents/provider-home.tsx"), - input.force === true, - created, - preserved, - overwritten - ); - writeProjectFile( - projectRoot, - ".smithers/agents/strict-json.ts", - loadRuntimeTemplate("smithers/agents/strict-json.tsx"), - input.force === true, - created, - preserved, - overwritten - ); - for (const agent of AGENT_TEMPLATES) { - const relativePath = `.smithers/agents/${agent.file}`; - if (upgradedStockAdapters.has(relativePath)) continue; - writeProjectFile( - projectRoot, - relativePath, - loadRuntimeTemplate(agent.template), - input.force === true, - created, - preserved, - overwritten - ); + // Planning admits only the byte-exact packaged adapter closure, so these + // files are never project-owned. Refresh them on every init: otherwise an + // upgrade needs `init --force`, which also resets ultrafuzz.toml, the + // topology, and the prompts. A file that already matches is left alone, + // so a read-only up-to-date closure does not fail init. + for (const [file, template] of Object.entries(STOCK_CONTROLLER_SOURCE_TEMPLATES)) { + const relativePath = `.smithers/agents/${file}`; + const contents = loadRuntimeTemplate(template); + if (holdsExactContents(projectRoot, relativePath, contents)) { + preserved.push(relativePath); + continue; + } + writeProjectFile(projectRoot, relativePath, contents, true, created, preserved, overwritten); } } catch { return runtimeFailure([ @@ -304,16 +195,12 @@ export function initProject(input: InitProjectInput) { preserved.push(toProjectRelative(projectRoot, absolutePath)); } - return runtimeResult( - true, - { - project_root: projectRoot, - created: publicInitPaths(created), - preserved: publicInitPaths(preserved), - overwritten: publicInitPaths(overwritten) - }, - [...staleAgentRegistryDiagnostics(projectRoot), ...staleAgentAdapterDiagnostics(projectRoot)] - ); + return runtimeResult(true, { + project_root: projectRoot, + created: publicInitPaths(created), + preserved: publicInitPaths(preserved), + overwritten: publicInitPaths(overwritten) + }); } function prepareStockSmithers032PackageMigration(projectRoot: string): string | undefined { @@ -340,141 +227,21 @@ function prepareStockSmithers032PackageMigration(projectRoot: string): string | } } -function upgradeStockSmithers032Adapters( - projectRoot: string, - created: string[], - preserved: string[], - overwritten: string[] -): ReadonlySet { - const upgraded = new Set(); - for (const agent of AGENT_TEMPLATES) { - const relativePath = `.smithers/agents/${agent.file}`; - const filePath = path.join(projectRoot, relativePath); - let bytes: Buffer; - try { - bytes = readStableInitReviewFile( - projectRoot, - filePath, - MAX_AGENT_ADAPTER_REVIEW_BYTES, - "generated 0.32 agent adapter" - ); - } catch { - continue; - } - const digest = crypto.createHash("sha256").update(bytes).digest("hex"); - if (!("stock032Sha256" in agent)) continue; - const stock032Sha256 = agent.stock032Sha256 as readonly string[]; - if (!stock032Sha256.includes(digest)) continue; - writeProjectFile( +/** Whether the path is a physical single-link file holding exactly these bytes. */ +function holdsExactContents(projectRoot: string, relativePath: string, contents: string): boolean { + const expected = Buffer.from(contents, "utf8"); + try { + return readStableInitReviewFile( projectRoot, - relativePath, - loadRuntimeTemplate(agent.template), - true, - created, - preserved, - overwritten - ); - upgraded.add(relativePath); - } - return upgraded; -} - -function staleAgentAdapterDiagnostics(projectRoot: string): RuntimeDiagnostic[] { - const diagnostics: RuntimeDiagnostic[] = []; - for (const agent of AGENT_TEMPLATES) { - const relativePath = `.smithers/agents/${agent.file}`; - const filePath = path.join(projectRoot, relativePath); - try { - const lexical = fs.lstatSync(filePath, { bigint: true }); - if (lexical.isSymbolicLink() || !lexical.isFile() || lexical.nlink !== 1n) { - diagnostics.push( - manualAgentAdapterReviewDiagnostic( - relativePath, - "is not a physical single-link file, so init preserved it without inspection; replace it with an ordinary file" - ) - ); - continue; - } - if (lexical.size > BigInt(MAX_AGENT_ADAPTER_REVIEW_BYTES)) { - diagnostics.push( - manualAgentAdapterReviewDiagnostic( - relativePath, - "is too large to inspect as a generated adapter and was preserved" - ) - ); - continue; - } - const source = readStableInitReviewFile( - projectRoot, - filePath, - MAX_AGENT_ADAPTER_REVIEW_BYTES, - "generated agent adapter" - ).toString("utf8"); - if (source.includes("ultrafuzz.toml") && !source.includes("ULTRAFUZZ_CONFIG_PATH")) { - diagnostics.push({ - code: "INIT_AGENT_ADAPTER_UPDATE_REQUIRED", - message: `${relativePath} was preserved and still reads mutable project ultrafuzz.toml; update it to read process.env.ULTRAFUZZ_CONFIG_PATH and use workflowControlChildEnvironment before spawning a model process`, - severity: "warning", - source: "runtime", - path: relativePath - }); - } - } catch (error) { - if (isNodeError(error) && error.code === "ENOENT") continue; - diagnostics.push( - manualAgentAdapterReviewDiagnostic(relativePath, "could not be safely inspected during post-init review") - ); - } - } - return diagnostics; -} - -function manualAgentAdapterReviewDiagnostic(relativePath: string, reason: string): RuntimeDiagnostic { - return { - code: "INIT_AGENT_ADAPTER_UPDATE_REQUIRED", - message: `${relativePath} ${reason}; verify manually that it reads process.env.ULTRAFUZZ_CONFIG_PATH and removes controller-only variables before spawning a model process`, - severity: "warning", - source: "runtime", - path: relativePath - }; -} - -// init preserves project-owned files, so a project scaffolded before an agent -// was added keeps its old registry: the new adapter lands on disk but nothing -// exports it, and the agent is only rejected later, at launch. Report it here -// instead of leaving the mismatch silent. -function staleAgentRegistryDiagnostics(projectRoot: string): RuntimeDiagnostic[] { - const registry = inspectAgentRegistry(projectRoot); - if (!registry.exists) return []; - if (registry.error !== undefined) { - // This runs after init may already have written other project files. Keep - // the warning actionable without reflecting an OS/parser error that can - // contain sensitive path or injected error details. - return [ - manualAgentRegistryReviewDiagnostic("was preserved without inspection because it could not be safely inspected") - ]; + path.join(projectRoot, relativePath), + expected.byteLength, + "generated agent adapter" + ).equals(expected); + } catch { + // Missing, linked, special, or larger files are handed to the writer, + // which creates the file or rejects the unsafe path. + return false; } - return AGENT_TEMPLATES.filter( - (agent) => - lstatIfPresent(path.join(projectRoot, ".smithers", "agents", agent.file)) !== undefined && - !agentRegistryRegisters(registry, agent.ref) - ).map((agent) => ({ - code: "INIT_AGENT_REGISTRY_STALE", - message: `${AGENT_REGISTRY_RELATIVE_PATH} does not register ${agent.ref} in agentFactories, so runs cannot select it; rerun ultrafuzz init --force to regenerate the registry, or add the entry by hand`, - severity: "warning" as const, - source: "runtime", - path: AGENT_REGISTRY_RELATIVE_PATH - })); -} - -function manualAgentRegistryReviewDiagnostic(reason: string): RuntimeDiagnostic { - return { - code: "INIT_AGENT_REGISTRY_REVIEW_REQUIRED", - message: `${AGENT_REGISTRY_RELATIVE_PATH} ${reason}; verify manually that agentFactories registers every generated agent before starting a run`, - severity: "warning", - source: "runtime", - path: AGENT_REGISTRY_RELATIVE_PATH - }; } function writeProjectFile( @@ -547,7 +314,9 @@ function writeProjectFileNoFollow( expected === undefined ? fs.constants.O_WRONLY | fs.constants.O_CREAT | fs.constants.O_EXCL | (fs.constants.O_NOFOLLOW ?? 0) : fs.constants.O_WRONLY | (fs.constants.O_NOFOLLOW ?? 0); - fileDescriptor = fs.openSync(accessPath, flags, 0o666); + // O_NONBLOCK makes a FIFO planted at a generated path fail the open (or the + // regular-file check below) instead of blocking init on a missing reader. + fileDescriptor = fs.openSync(accessPath, flags | fs.constants.O_NONBLOCK, 0o666); const opened = fs.fstatSync(fileDescriptor, { bigint: true }); // Modal's virtual filesystem can report one device for an opened // directory and another for stable children created through that dirfd. @@ -714,10 +483,6 @@ function writeDescriptorContents(descriptor: number, contents: Buffer): void { } } -function isNodeError(error: unknown): error is NodeJS.ErrnoException { - return error instanceof Error && "code" in error; -} - function uniqueSorted(values: string[]): string[] { return Array.from(new Set(values)).sort(); } diff --git a/packages/runtime/src/lifecycle-inspection.ts b/packages/runtime/src/lifecycle-inspection.ts index 3aa38ac59..cb81f78a6 100644 --- a/packages/runtime/src/lifecycle-inspection.ts +++ b/packages/runtime/src/lifecycle-inspection.ts @@ -237,7 +237,13 @@ function readRunStatusIfPresent(layout: RunLayout): RunStatus | undefined { export async function cancelRun(input: CancelRunInput) { const projectRoot = path.resolve(input.projectRoot); - const evidence = await readLinkedWorkflowEvidence(projectRoot, input.runId); + // Cancelling runs none of the run's workflow code: it only asks the workflow runner to stop the + // linked run. So it reads evidence the way `status` does, and a run whose sealed control documents + // diverged can still be stopped instead of refusing with WORKFLOW_CONTROL_EVIDENCE_INVALID. + const evidence = await readLinkedWorkflowEvidence(projectRoot, input.runId, { + tolerateControlDivergence: true, + observeOnly: true + }); if (!evidence.ok) { return runtimeFailure(evidence.diagnostics); } diff --git a/packages/runtime/src/plan-run.ts b/packages/runtime/src/plan-run.ts index 23a091dd0..1e0439fa0 100644 --- a/packages/runtime/src/plan-run.ts +++ b/packages/runtime/src/plan-run.ts @@ -43,7 +43,7 @@ import { type PromptConcreteNode, type PromptGraphNode } from "@ultrafuzz/prompts"; -import { loadReferenceCatalog, materializeReferenceArtifacts } from "@ultrafuzz/references"; +import { loadReferenceCatalog, materializeReferenceArtifacts, verifyReferencesCached } from "@ultrafuzz/references"; import { expandTopology, fingerprintGraph, @@ -108,6 +108,8 @@ interface PlanRunHooks { resolvedConfig: PlanRunValue["resolved_config"]; expandedGraph: ExpandedGraph; }): Promise; + /** The run directory now exists, so a later planning failure leaves it for the caller to record. */ + afterLayoutCreated?(layout: RunLayout): void; } export async function planRun(input: PlanRunInput, hooks: PlanRunHooks = {}) { @@ -130,6 +132,9 @@ export async function planRun(input: PlanRunInput, hooks: PlanRunHooks = {}) { if (!validation.ok || !validation.value) { return runtimeFailure(validation.diagnostics); } + if (validation.value.topology === undefined) { + throw new Error("validated run plan is missing its topology summary"); + } const resolved = await loadResolvedProject(input); if (!resolved.config) { @@ -228,6 +233,35 @@ export async function planRun(input: PlanRunInput, hooks: PlanRunHooks = {}) { if (hasRuntimeErrors(graphDiagnostics)) { return runtimeFailure(graphDiagnostics); } + // Graph-only checks belong before the run directory exists, so an invalid topology commits nothing. + const requiresVulnerabilityDatabase = graph.nodes.some( + (node) => node.logical_id === "threat-model" || node.logical_id === "goal-plan" + ); + const vulnerabilityDatabaseReferenceNode = graph.nodes.find( + (node) => node.logical_id === VULNERABILITY_DATABASE_REFERENCE_NODE_ID && node.kind === "reference" + ); + if (requiresVulnerabilityDatabase && vulnerabilityDatabaseReferenceNode === undefined) { + return runtimeFailure([ + { + code: "VULNERABILITY_DATABASE_REFERENCE_REQUIRED", + message: `topology nodes threat-model/goal-plan require the ${VULNERABILITY_DATABASE_REFERENCE_NODE_ID} pinned reference node`, + severity: "error", + source: "vulnerability-database" + } + ]); + } + const referenceIds = graph.nodes.flatMap((node) => + node.kind === "reference" && node.reference !== undefined ? [node.reference] : [] + ); + if (referenceIds.length > 0) { + // Materialization reads these caches after the run directory exists. Checking them here, before + // governance and provider preflight, means a missing or stale cache leaves no run behind. + try { + verifyReferencesCached(loadReferenceCatalog(projectRoot), referenceIds); + } catch (error) { + return runtimeFailure([diagnosticFromError(error, "references", "REFERENCE_MATERIALIZE_FAILED")]); + } + } let controllerSource: ReturnType; try { controllerSource = inspectControllerSource(projectRoot); @@ -382,6 +416,7 @@ export async function planRun(input: PlanRunInput, hooks: PlanRunHooks = {}) { forge_guard: forgeGuardMetadata(resolved.config, false) } }); + hooks.afterLayoutCreated?.(layout); writeFileDurable(path.join(layout.root, DATA_GOVERNANCE_PROVENANCE_PATH), governanceBytes); } catch (error) { return runtimeFailure([diagnosticFromError(error, "runtime", "RUN_LAYOUT_INVALID")]); @@ -395,27 +430,11 @@ export async function planRun(input: PlanRunInput, hooks: PlanRunHooks = {}) { } let vulnerabilityDatabase: MaterializedVulnerabilityDatabaseCatalog | undefined; - const requiresVulnerabilityDatabase = graph.nodes.some( - (node) => node.logical_id === "threat-model" || node.logical_id === "goal-plan" - ); - if (requiresVulnerabilityDatabase) { - const referenceNode = graph.nodes.find( - (node) => node.logical_id === VULNERABILITY_DATABASE_REFERENCE_NODE_ID && node.kind === "reference" - ); - if (referenceNode === undefined) { - return runtimeFailure([ - { - code: "VULNERABILITY_DATABASE_REFERENCE_REQUIRED", - message: `topology nodes threat-model/goal-plan require the ${VULNERABILITY_DATABASE_REFERENCE_NODE_ID} pinned reference node`, - severity: "error", - source: "vulnerability-database" - } - ]); - } + if (requiresVulnerabilityDatabase && vulnerabilityDatabaseReferenceNode !== undefined) { try { vulnerabilityDatabase = materializeVulnerabilityDatabasePlannerCatalog( layout.root, - getNodeArtifactDir(layout, referenceNode.id) + getNodeArtifactDir(layout, vulnerabilityDatabaseReferenceNode.id) ); } catch (error) { return runtimeFailure([ @@ -440,15 +459,13 @@ export async function planRun(input: PlanRunInput, hooks: PlanRunHooks = {}) { return runtimeFailure([diagnosticFromError(error, "prompts", "PROMPT_RENDER_FAILED")]); } - const persistedRenderedPrompts = persistRenderedPromptSnapshots(layout, renderedPrompts); + let persistedRenderedPrompts: ReturnType; try { + persistedRenderedPrompts = persistRenderedPromptSnapshots(layout, renderedPrompts); persistDeferredPromptTemplates(layout, catalog, expandedGraph); } catch (error) { return runtimeFailure([diagnosticFromError(error, "prompts", "PROMPT_TEMPLATE_SNAPSHOT_FAILED")]); } - if (validation.value.topology === undefined) { - throw new Error("validated run plan is missing its topology summary"); - } if (sourceRevision !== undefined) { try { assertLaunchCheckoutRevision(projectRoot, sourceRevision); diff --git a/packages/runtime/src/runtime-document-codec.ts b/packages/runtime/src/runtime-document-codec.ts index 29e116c2c..48fd87106 100644 --- a/packages/runtime/src/runtime-document-codec.ts +++ b/packages/runtime/src/runtime-document-codec.ts @@ -25,7 +25,11 @@ export function parseRuntimeDocumentBytes { - if (!isRecord(value)) { - return [ - { - code: "SEVERITY_RECORD_SHAPE_INVALID", - message: `${path} must be an object`, - severity: "error" as const, - source: "severity-matrix", - path: `${input.artifactPath}#/${path}` - } - ]; - } - return validateRecord(value, path, input.artifactPath); - }); -} - -function recordsForArtifact( - artifact: unknown, - kind: SeverityArtifactKind -): Array<{ value: unknown; path: string }> | undefined { - if (kind === "final-report") { - if (!isRecord(artifact) || !Array.isArray(artifact.issues)) { - return undefined; - } - return artifact.issues.map((value, index) => ({ value, path: `issues.${index}` })); - } - if (!Array.isArray(artifact)) { - return undefined; - } - return artifact.map((value, index) => ({ value, path: `${index}` })); -} - -function validateRecord( - record: Record, - recordPath: string, - artifactPath: string -): RuntimeDiagnostic[] { - const diagnostics: RuntimeDiagnostic[] = []; - - for (const field of LEGACY_FIELDS) { - if (field in record) { - diagnostics.push({ - code: "SEVERITY_FIELD_ALIAS_UNSUPPORTED", - message: `${recordPath}.${field} is not supported; use the canonical severity, impact, and likelihood fields`, - severity: "error", - source: "severity-matrix", - path: `${artifactPath}#/${recordPath}.${field}` - }); - } - } - - const severity = validateLevel(record, "severity", recordPath, artifactPath, diagnostics); - const impact = validateLevel(record, "impact", recordPath, artifactPath, diagnostics); - const likelihood = validateLevel(record, "likelihood", recordPath, artifactPath, diagnostics); - const expected = expectedSeverityFromMatrix(impact, likelihood); - if (severity !== undefined && expected !== undefined && severity !== expected) { - diagnostics.push({ - code: "SEVERITY_MATRIX_MISMATCH", - message: `${recordPath} has severity ${severity}, impact ${impact}, likelihood ${likelihood}; expected ${expected}`, - severity: "error", - source: "severity-matrix", - path: `${artifactPath}#/${recordPath}` - }); - } - - return diagnostics; -} - -function validateLevel( - record: Record, - field: "severity" | "impact" | "likelihood", - recordPath: string, - artifactPath: string, - diagnostics: RuntimeDiagnostic[] -): SeverityLevel | undefined { - const value = record[field]; - if (value === undefined || value === null) { - diagnostics.push({ - code: "SEVERITY_MATRIX_FIELD_MISSING", - message: `${recordPath} has no machine-readable ${field}`, - severity: "error", - source: "severity-matrix", - path: `${artifactPath}#/${recordPath}` - }); - return undefined; - } - if (!isSeverityLevel(value)) { - diagnostics.push({ - code: "SEVERITY_LEVEL_INVALID", - message: `${recordPath}.${field} must be one of ${LEVELS.join(", ")}, got ${JSON.stringify(value)}`, - severity: "error", - source: "severity-matrix", - path: `${artifactPath}#/${recordPath}.${field}` - }); - return undefined; - } - return value; -} - -function isSeverityLevel(value: unknown): value is SeverityLevel { - return value === "High" || value === "Medium" || value === "Low"; -} - -function isRecord(value: unknown): value is Record { - return value !== null && typeof value === "object" && !Array.isArray(value); -} diff --git a/packages/runtime/src/smithers.ts b/packages/runtime/src/smithers.ts index 51c96f889..a7b3c9c9c 100644 --- a/packages/runtime/src/smithers.ts +++ b/packages/runtime/src/smithers.ts @@ -1,7 +1,7 @@ import { execFile, spawn } from "node:child_process"; import crypto from "node:crypto"; import fs from "node:fs"; -import { isBuiltin } from "node:module"; +import { createRequire, isBuiltin } from "node:module"; import os from "node:os"; import path from "node:path"; import { createInterface } from "node:readline"; @@ -9,7 +9,6 @@ import { fileURLToPath, pathToFileURL } from "node:url"; import { promisify } from "node:util"; import { - artifactContractDefinition, artifactContractSchemaBinding, assertRunPlanDocument, assertValidSmithersTaskManifest, @@ -60,7 +59,7 @@ import { type ExpandedNode, type ModelFanoutProvenance } from "@ultrafuzz/topology"; -import * as ts from "typescript"; +import type * as TypeScript from "typescript"; import { DATA_GOVERNANCE_PROVENANCE_PATH, @@ -69,6 +68,7 @@ import { routeOwnsCredentialLikeEnvironmentVariable } from "./data-governance.js"; import { archiveDynamicExpansionsForRetry, planDynamicExpansionRetryArchive } from "./dynamic-expansion-retry.js"; +import { reconcilesPartialResults } from "./dynamic-runtime.js"; import { assertControllerSourceDigest, inspectControllerSource, @@ -904,11 +904,16 @@ export async function readWorkflowGraphHash(workflowPath, identityWorkflowPath = workflowPath, identityWorkflowPath || workflowPath, );`; -// Restores exactly the two states upstream's `isTerminalState` calls terminal -// unconditionally: `finished` and `skipped`. `failed`, `cancelled` and Smithers -// 0.35.0's new `stalled` are deliberately NOT restored, and the omission of -// `stalled` is the deliberate half of that rule, not an oversight from the -// 0.35.0 bump. Upstream classes `stalled` with `failed` ("it behaves exactly +// Restores only `finished`, the one state backed by a durable output row. +// `skipped` is re-derived instead: it is a verdict on prerequisites that a reset +// (`resume --retry-failed`, `--reset-node`) can overturn, and restoring it kept a +// recovered producer's verifier and every descendant skipped (#1141). The flag +// set here makes the session re-render before it schedules anything (see +// `skip_predicate_rerender`), so the workflow's skip predicates see the +// restored states. +// `failed`, `cancelled` and Smithers 0.35.0's new `stalled` are deliberately NOT +// restored, and the omission of `stalled` is not an oversight from the 0.35.0 +// bump. Upstream classes `stalled` with `failed` ("it behaves exactly // like `failed`, including the continueOnFail escape hatch"), and a resume's // whole purpose is to re-attempt what did not finish -- restoring `stalled` but // not `failed` would make a stalled node strictly less retryable than an @@ -924,11 +929,42 @@ const SMITHERS_SCHEDULER_TERMINAL_RESTORE_SOURCE = const SMITHERS_SCHEDULER_TERMINAL_RESTORE_PATCH = ` restoreTerminalTaskStates: (tasks) => Effect.sync(() => { for (const task of tasks) { - if (task.state !== "finished" && task.state !== "skipped") continue; + if (task.state !== "finished") continue; state.states.set(stateKeyFor(task), task.state); } + state.skipPredicatesStale = true; }), getTaskStates: () => Effect.sync(() => cloneTaskStateMap(state.states)),`; +// Ultrafuzz's skip predicates read other nodes' states, so a predicate is only +// as current as the render that computed it. Upstream re-renders when a node +// finishes or exhausts its retries, but keeps scheduling from an older graph +// after two other state changes: +// - resume hydration: a resumed session's first graph was rendered before it, +// so a still-failed agent's verifier ran; +// - a skip: the dependents it made runnable kept predicates that never saw it, +// so each descendant of a failed agent ran its preparation into a failure. +// That happens on the recursive pass after a skip, and, when the skip shared +// its pass with dispatched work, on the next decision made without a render, +// such as after a retryable failure or at a retry deadline. +// Both set the flag, and the next decision re-renders before scheduling. That +// costs at most one redundant render per skipping pass, when a completion's own +// re-render already saw the skip. A pass that only skipped now ends in a +// re-render instead of recursing on the same graph, so upstream's decide() +// depth guard no longer counts it; those re-renders stay bounded because a +// session only returns a skipped node to pending through `hotReloaded`, which +// the pinned engine never calls. +const SMITHERS_SCHEDULER_SKIP_RERENDER_SOURCE = ` if (!state.graph) { + return { _tag: "Wait", reason: { _tag: "ExternalTrigger" } }; + }`; +const SMITHERS_SCHEDULER_SKIP_RERENDER_PATCH = `${SMITHERS_SCHEDULER_SKIP_RERENDER_SOURCE} + if (state.skipPredicatesStale) { + state.skipPredicatesStale = false; + return { _tag: "ReRender", context: renderContext(state, undefined, { reason: "skip-check" }) }; + }`; +const SMITHERS_SCHEDULER_SKIP_MARK_SOURCE = ` if (task.skipIf) { + state.states.set(key, "skipped");`; +const SMITHERS_SCHEDULER_SKIP_MARK_PATCH = `${SMITHERS_SCHEDULER_SKIP_MARK_SOURCE} + state.skipPredicatesStale = true;`; // Anchored immediately after the resume path's `startRunRuntime()`, which is // where Smithers cancels stale in-progress attempts and rewrites their nodes back // to `pending`. Hydrating before that reset would restore a node as finished and @@ -946,9 +982,6 @@ const SMITHERS_ENGINE_RESUME_HYDRATION_PATCH = ` resumeWorkflowNameVali const durableOutputs = await loadOutputs(db, schema, runId); const durableNodes = await Effect.runPromise(adapter.listNodes(runId)); const terminalTaskStates = durableNodes.flatMap((node) => { - if (node.state === "skipped") { - return [{ nodeId: node.nodeId, iteration: node.iteration ?? 0, state: "skipped" }]; - } if (node.state !== "finished" || typeof node.outputTable !== "string") return []; const rows = durableOutputs[node.outputTable]; const hasOutput = @@ -1011,6 +1044,63 @@ const SMITHERS_ENGINE_AGENT_EVENT_OWNERSHIP_PATCH = ` const pendingOwnershipChe pendingOwnershipChecks.add(check); };`; +// After each successful fenced attempt-row heartbeat write, the engine appends a +// TaskHeartbeat event (an `_smithers_events` row plus a stream.ndjson line) that +// carries no heartbeat data, because Ultrafuzz never passes any. A quiet agent +// task writes one per throttled liveness pulse, up to two a second; an agent that +// streams output writes one per ownership check its stdout, stderr and tool +// callbacks force past the throttle. In a baseline campaign they were 83% of the +// event rows (#1147), and Ultrafuzz never acts on them (it handles only +// TaskHeartbeatTimeout). Liveness is the attempt-row write: the heartbeat-timeout +// watchdog advances only when it succeeds, and `smithers why` reads the row. Keep +// the write and drop the event. +const SMITHERS_ENGINE_TASK_HEARTBEAT_EVENT_SOURCE = ` "heartbeat:record", + ); + await eventBus.emitEventQueued({ + type: "TaskHeartbeat", + runId, + nodeId: desc.nodeId, + iteration: desc.iteration, + attempt: attemptNo, + hasData: heartbeatDataJson !== null, + dataSizeBytes, + intervalMs: intervalMs ?? undefined, + timestampMs: heartbeatAtMs, + }); + } catch (error) {`; +const SMITHERS_ENGINE_TASK_HEARTBEAT_EVENT_PATCH = ` "heartbeat:record", + ); + // ultrafuzz: the fenced attempt row above is the liveness record (#1147). + } catch (error) {`; + +// Smithers treats as a branch to track. Creating a +// worktree runs `git fetch origin` first, and each re-entry retries +// `git rebase origin/` until one succeeds, after another fetch unless one +// succeeded for that repository in the last 60 s. Ultrafuzz passes the recorded +// launch commit (or a pinned source branch), and `origin/` never resolves, +// so every re-entry logs a failed rebase; each fetch is an untimed call to the +// user's remote (#1148). Task worktrees must stay on the launch commit +// (assertWorkspaceSourceRevision), so never synchronize them. +const SMITHERS_ENGINE_WORKTREE_SYNC_SOURCE = `function getWorktreeSyncCache() { + if (!worktreeSyncCacheSingleton) { + worktreeSyncCacheSingleton = createWorktreeSyncCache({ ttlMs: resolveWorktreeFetchTtlMs() }); + } + return worktreeSyncCacheSingleton; +}`; +const SMITHERS_ENGINE_WORKTREE_SYNC_PATCH = `function getWorktreeSyncCache() { + // ultrafuzz: task worktrees stay on their recorded launch commit (#1148). + return { shouldFetch: () => false, recordFetch() {}, shouldRebase: () => false, recordRebase() {} }; +}`; +const SMITHERS_ENGINE_WORKTREE_CREATE_FETCH_SOURCE = ` // Best effort: refresh remote refs for git so origin/main can be used as a + // base when local main is absent. + if (vcs.type === "git") { + await runGitCommand(vcs.root, ["fetch", "origin"]); + } +`; +const SMITHERS_ENGINE_WORKTREE_CREATE_FETCH_PATCH = ` // ultrafuzz: task worktrees start from a local recorded commit, so creating + // one never fetches origin (#1148). +`; + // Every event the engine persists first runs an idempotency probe that // filters `_smithers_events` on (run_id, timestamp_ms, type, payload_json). // The table's only index is its (run_id, seq) primary key, and the probe's @@ -2436,8 +2526,13 @@ export type SmithersCompatibilityPatchId = | "supervisor_descriptor" | "resume_snapshot_transfer" | "terminal_state_restore" + | "skip_predicate_rerender" + | "skip_marks_predicates_stale" | "resume_hydration" | "engine_agent_event_ownership" + | "engine_task_heartbeat_event" + | "engine_worktree_sync" + | "engine_worktree_create_fetch" | "engine_agent_usage_progress" | "engine_main_usage_invocation" | "engine_json_correction_usage_invocation" @@ -2567,6 +2662,25 @@ export const SMITHERS_COMPATIBILITY_PATCHES: readonly SmithersCompatibilityPatch // Upstream growing its own terminal-state restoration retires this patch. upstreamAbsent: ["restoreTerminalTaskStates"] }, + { + id: "skip_predicate_rerender", + packageName: "@smthrs/scheduler", + sourceRelativePath: "src/makeWorkflowSession.js", + patchable: SMITHERS_SCHEDULER_SKIP_RERENDER_SOURCE, + patched: SMITHERS_SCHEDULER_SKIP_RERENDER_PATCH, + // Consumes the flag that `terminal_state_restore` and + // `skip_marks_predicates_stale` set, so it retires only with both. + upstreamAbsent: [] + }, + { + id: "skip_marks_predicates_stale", + packageName: "@smthrs/scheduler", + sourceRelativePath: "src/makeWorkflowSession.js", + patchable: SMITHERS_SCHEDULER_SKIP_MARK_SOURCE, + patched: SMITHERS_SCHEDULER_SKIP_MARK_PATCH, + // Retires once upstream re-renders after a skip; no upstream text names that yet. + upstreamAbsent: [] + }, { id: "resume_hydration", packageName: "@smthrs/engine", @@ -2585,6 +2699,30 @@ export const SMITHERS_COMPATIBILITY_PATCHES: readonly SmithersCompatibilityPatch // Upstream coalescing its own in-flight proof retires this patch. upstreamAbsent: ["heartbeatOwnershipCheckInFlight"] }, + { + id: "engine_task_heartbeat_event", + packageName: "@smthrs/engine", + sourceRelativePath: "src/engine.js", + patchable: SMITHERS_ENGINE_TASK_HEARTBEAT_EVENT_SOURCE, + patched: SMITHERS_ENGINE_TASK_HEARTBEAT_EVENT_PATCH, + upstreamAbsent: [] + }, + { + id: "engine_worktree_sync", + packageName: "@smthrs/engine", + sourceRelativePath: "src/engine.js", + patchable: SMITHERS_ENGINE_WORKTREE_SYNC_SOURCE, + patched: SMITHERS_ENGINE_WORKTREE_SYNC_PATCH, + upstreamAbsent: [] + }, + { + id: "engine_worktree_create_fetch", + packageName: "@smthrs/engine", + sourceRelativePath: "src/engine.js", + patchable: SMITHERS_ENGINE_WORKTREE_CREATE_FETCH_SOURCE, + patched: SMITHERS_ENGINE_WORKTREE_CREATE_FETCH_PATCH, + upstreamAbsent: [] + }, { id: "engine_agent_usage_progress", packageName: "@smthrs/engine", @@ -3149,6 +3287,8 @@ export interface CurrentSmithersInspect { nodes: CurrentSmithersInspectNode[]; failedChildKeys: string[]; exhaustedLoops: CurrentSmithersExhaustedLoop[]; + /** The runner's run-level error (for example WORKFLOW_RENDER_FAILED), redacted and capped. Advisory text only. */ + runError?: { code?: string; message: string }; } /** @@ -3746,6 +3886,8 @@ function assertRefreshedModuleAuthority( } const dependencies = new Set(Object.keys(issuers[0]!.dependencies)); const executablePaths = new Set(dependencyMap.executable_paths); + // Required here rather than imported: every ultrafuzz CLI process imports this module. + const ts = createRequire(import.meta.url)("typescript") as typeof TypeScript; for (const file of files) { if (file.executable !== executablePaths.has(file.snapshotPath)) { throw new Error(`controller module ${moduleName} changed executable authority for ${file.snapshotPath}`); @@ -4077,12 +4219,9 @@ export function compileSmithersWorkflow(input: SmithersCompileInput): CompiledSm const nonBlockingAttemptIdSet = new Set(nonBlockingAttemptIds); const tasks = compiledTasks.map((task) => ({ ...task, - // Continuation lets independent tasks settle. It does not make a strategy's - // required inputs optional; only the review group reconciles partial results. - optionalDependencyArtifactDirs: - task.metadata.node.group === "review" - ? task.dependencyArtifactDirs.filter((directory) => nonBlockingAttemptIdSet.has(path.basename(directory))) - : [] + optionalDependencyArtifactDirs: reconcilesPartialResults(task) + ? task.dependencyArtifactDirs.filter((directory) => nonBlockingAttemptIdSet.has(path.basename(directory))) + : [] })); const smithersDir = path.join(input.runLayout.root, "smithers"); fs.mkdirSync(smithersDir, { recursive: true }); @@ -4337,12 +4476,6 @@ export async function smithersExecutionControlFiles( const agentsRoot = path.join(compiled.projectRoot, ".smithers", "agents"); assertControllerSourceDigest(compiled.projectRoot, compiled.controllerSourceDigest); for (const sourcePath of walkExecutionFiles(agentsRoot)) { - const source = fs.readFileSync(sourcePath, "utf8"); - if (source.includes("ultrafuzz.toml") && !source.includes("ULTRAFUZZ_CONFIG_PATH")) { - throw new Error( - `workflow agent reads mutable project ultrafuzz.toml instead of process.env.ULTRAFUZZ_CONFIG_PATH: ${sourcePath}; rerun ultrafuzz init --force to replace generated files, or update this adapter manually and remove controller-only variables before spawning a model process` - ); - } add(sourcePath, path.posix.join(".smithers/agents", relativeExecutionPath(agentsRoot, sourcePath))); } @@ -5209,8 +5342,6 @@ export async function runSmithersLifecycleCommand(input: { priorInspection?: SmithersResumeInspection; /** Prepare launch authority only after ruling out an idempotent active attach. */ prepareContinuationEnvironment?: () => Record; - /** Preserve stopped-run failure evidence before its mutable attempt rows are reset. */ - beforeStoppedReset?: (inspection: SmithersCommandSnapshot) => Promise; relaunchPaths?: { runRoot: string; inputJson?: string; @@ -5256,18 +5387,6 @@ export async function runSmithersLifecycleCommand(input: { let preResumeStderr = ""; let currentInspection: CurrentSmithersInspect | undefined; let inspection: SmithersCommandSnapshot | undefined; - let stoppedResetPreserved = false; - const preserveStoppedReset = async (): Promise => { - if ( - stoppedResetPreserved || - currentInspection === undefined || - inspection === undefined || - smithersRunStateIsActive(currentInspection) - ) - return; - await input.beforeStoppedReset?.(inspection); - stoppedResetPreserved = true; - }; // Detached admission renders the workflow before Smithers checks whether // this run already has an active owner. Inspect every resume first so an // idempotent attach cannot fail preflight or compete with that owner (#968). @@ -5336,7 +5455,6 @@ export async function runSmithersLifecycleCommand(input: { }) : undefined; if (failedTasks.length > 0) { - await preserveStoppedReset(); const resetStderr: string[] = []; for (const failedTask of failedTasks) { const producerTask = retryProducerForFailedVerifier(currentInspection, failedTask); @@ -5390,7 +5508,6 @@ export async function runSmithersLifecycleCommand(input: { : path.join(input.relaunchPaths.runRoot, "smithers", "reset-node-applied.json"); let resetStderr = ""; if (!resetNodeMarkerMatches(resetMarkerPath, input.smithersRunId, input.resetNode)) { - await preserveStoppedReset(); // A failure can be durable in the canonical node snapshot even when the // runner cannot resolve its implicit "latest attempt" lookup. Pinning the // iteration from that snapshot keeps --reset-node recoverable by node ID. @@ -5635,6 +5752,20 @@ export async function assertSmithersControllerRefreshable(input: { return { status: "present", snapshot: inspection, inspect }; } +const DEFAULT_RUNNER_QUERY_TIMEOUT_MS = 120_000; +const MAX_RUNNER_QUERY_TIMEOUT_MS = 600_000; + +/** + * Bounds one read-only runner query, so a wedged runner becomes a failed snapshot instead of blocking + * `status`, `inspect`, `stats`, or a synchronization pump forever. `ULTRAFUZZ_RUNNER_QUERY_TIMEOUT_MS` + * overrides the default with a positive number of milliseconds, capped at ten minutes. + */ +function runnerQueryTimeoutMs(env: Record | undefined): number { + const configured = (env ?? process.env).ULTRAFUZZ_RUNNER_QUERY_TIMEOUT_MS; + const parsed = configured !== undefined && /^[1-9]\d*$/u.test(configured) ? Number(configured) : Number.NaN; + return Number.isSafeInteger(parsed) ? Math.min(parsed, MAX_RUNNER_QUERY_TIMEOUT_MS) : DEFAULT_RUNNER_QUERY_TIMEOUT_MS; +} + export async function runSmithersInspectionCommand(input: { args: readonly string[]; projectRoot: string; @@ -5644,6 +5775,7 @@ export async function runSmithersInspectionCommand(input: { timeoutMs?: number; }): Promise { const command = [...input.args]; + const commandTimeoutMs = runnerQueryTimeoutMs(input.env); try { const result = await execSmithersCli({ args: command, @@ -5651,7 +5783,8 @@ export async function runSmithersInspectionCommand(input: { env: input.env, environmentVariableNames: input.environmentVariableNames, signal: input.signal, - timeoutMs: input.timeoutMs + timeoutMs: input.timeoutMs, + commandTimeoutMs }); return { command: result.command, @@ -5662,16 +5795,24 @@ export async function runSmithersInspectionCommand(input: { }; } catch (error) { const record = - error && typeof error === "object" ? (error as { stdout?: unknown; stderr?: unknown; message?: unknown }) : {}; + error && typeof error === "object" + ? (error as { stdout?: unknown; stderr?: unknown; message?: unknown; killed?: unknown; signal?: unknown }) + : {}; const stdout = typeof record.stdout === "string" ? record.stdout : ""; const stderr = typeof record.stderr === "string" ? record.stderr : ""; + // Node reports the kill of its own execution timeout as `killed` with SIGTERM. + const timedOut = record.killed === true && record.signal === "SIGTERM"; return { command: smithersDisplayCommand(command), ok: false, stdout, stderr, ...jsonField(stdout), - error: error instanceof Error ? error.message : String(error) + error: timedOut + ? `workflow runner query exceeded its time limit and was stopped (ULTRAFUZZ_RUNNER_QUERY_TIMEOUT_MS=${String(commandTimeoutMs)})` + : error instanceof Error + ? error.message + : String(error) }; } } @@ -5965,7 +6106,41 @@ export function parseCurrentSmithersInspect( if (exhaustedLoops.length > 0 && parsedRunState !== "succeeded" && parsedRunState !== "succeeded-with-failures") { throw new Error("Smithers inspect data.exhaustedLoops is only valid for a succeeded workflow state"); } - return { runStatus, runState: parsedRunState, nodes, failedChildKeys, exhaustedLoops }; + const runError = currentSmithersRunError(run.error); + return { + runStatus, + runState: parsedRunState, + nodes, + failedChildKeys, + exhaustedLoops, + ...(runError === undefined ? {} : { runError }) + }; +} + +// A run-level failure (for example a render exception) names no task, so the +// run row's error is the only record of why the run stopped. Read it loosely: +// an unexpected shape yields nothing and never fails the parse. +function currentSmithersRunError(value: unknown): CurrentSmithersInspect["runError"] { + if (!isObjectRecord(value)) return undefined; + const nonBlank = (field: unknown): field is string => typeof field === "string" && field.trim() !== ""; + // The runner records what was thrown as `cause`. Its `summary` is its own + // message before it appends a docs link and raw runner resume commands. + const cause = isObjectRecord(value.cause) ? value.cause.message : undefined; + const message = [cause, value.summary, value.message].find(nonBlank); + if (message === undefined) return undefined; + return { + ...(nonBlank(value.code) ? { code: runErrorText(value.code, 100) } : {}), + message: runErrorText(message, 1_000) + }; +} + +// Redacted and scrubbed like other runner text, since it is printed. The runner +// stores error text untruncated and redaction cost grows with the square of one +// long token, so only a prefix is redacted. For the message, eight times the +// kept length still holds a whole PEM private key (3,300 characters at RSA-4096) +// that starts in the kept text, so it is redacted as one block. +function runErrorText(value: string, limit: number): string { + return scrubWorkflowRunnerText(redactSecretsInText(value.slice(0, limit * 8))).slice(0, limit); } function parseCurrentSmithersExhaustedLoops(value: unknown): CurrentSmithersExhaustedLoop[] { @@ -6164,32 +6339,6 @@ function requiredCurrentInspectEnum( return value as Values[number]; } -/** - * Dependency attempt ids whose verified artifacts no longer match their verification marker. - * The message is emitted by the generated workflow's own `assertVerifiedDependency`, so the shape - * is stable, and it is the only signal that reaches the resume side: the failure lands on the - * dependent's `prepare:` task and leaves no failed node behind, so the run row's `error_json` is - * where it surfaces. Recovery on top of this is tracked separately in #288. - */ -export function smithersSnapshotUnverifiedDependencies(snapshot: SmithersCommandSnapshot): string[] { - const evidence = [ - snapshot.stdout, - snapshot.stderr, - snapshot.error ?? "", - snapshot.json === undefined ? "" : JSON.stringify(snapshot.json) - ].join("\n"); - const dependencies = new Set(); - for (const match of evidence.matchAll( - /artifact dependency has not passed verification ([A-Za-z0-9._-]+) for [A-Za-z0-9._-]+/gu - )) { - const dependency = match[1]; - if (dependency !== undefined && dependency.trim() !== "" && !dependency.includes("..")) { - dependencies.add(dependency); - } - } - return [...dependencies].sort(); -} - function isCompatibleSmithersRunId(value: string): boolean { return /^[a-z0-9_-]{1,64}$/u.test(value); } @@ -6236,6 +6385,8 @@ async function execSmithersCli(input: { acceptedExitCodes?: readonly number[]; signal?: AbortSignal; timeoutMs?: number; + /** Bounds only the runner process, never executable preparation. */ + commandTimeoutMs?: number; }): Promise<{ stdout: string; stderr: string; command: string[]; exitCode: number }> { const command = [...input.args]; const executionDeadline = input.timeoutMs === undefined ? undefined : Date.now() + input.timeoutMs; @@ -6243,8 +6394,10 @@ async function execSmithersCli(input: { signal: input.signal, timeoutMs: input.timeoutMs }); - const commandTimeoutMs = + const remainingMs = executionDeadline === undefined ? undefined : Math.max(1, Math.ceil(executionDeadline - Date.now())); + const commandTimeoutMs = + remainingMs === undefined ? input.commandTimeoutMs : Math.min(remainingMs, input.commandTimeoutMs ?? remainingMs); const { anchored, snapshotAnchor, executableAnchor } = acquireAnchoredSmithersController(command, commandEnvironment); try { const { stdout, stderr } = await execFileAsync( @@ -6990,16 +7143,19 @@ export function applySmithersCompatibilityPatches(projectRoot: string): void { } assertRegularFileInside(nodeModules, engineWorkflowHashSource, "installed Smithers workflow hash implementation"); - const schedulerContents = fs.readFileSync(schedulerSource, "utf8"); - writeFileDurable( - schedulerSource, - applyRequiredSmithersPatch( - schedulerContents, + let schedulerContents = fs.readFileSync(schedulerSource, "utf8"); + for (const [source, patched, label] of [ + [ SMITHERS_SCHEDULER_TERMINAL_RESTORE_SOURCE, SMITHERS_SCHEDULER_TERMINAL_RESTORE_PATCH, "terminal-state restoration" - ) - ); + ], + [SMITHERS_SCHEDULER_SKIP_RERENDER_SOURCE, SMITHERS_SCHEDULER_SKIP_RERENDER_PATCH, "skip predicate re-render"], + [SMITHERS_SCHEDULER_SKIP_MARK_SOURCE, SMITHERS_SCHEDULER_SKIP_MARK_PATCH, "skip marks predicates stale"] + ] as const) { + schedulerContents = applyRequiredSmithersPatch(schedulerContents, source, patched, label); + } + writeFileDurable(schedulerSource, schedulerContents); let workflowHashContents = fs.readFileSync(engineWorkflowHashSource, "utf8"); for (const [source, patched, label] of [ @@ -7067,6 +7223,13 @@ export function applySmithersCompatibilityPatches(projectRoot: string): void { SMITHERS_ENGINE_AGENT_EVENT_OWNERSHIP_PATCH, "agent event ownership coalescing" ], + [SMITHERS_ENGINE_TASK_HEARTBEAT_EVENT_SOURCE, SMITHERS_ENGINE_TASK_HEARTBEAT_EVENT_PATCH, "task heartbeat event"], + [SMITHERS_ENGINE_WORKTREE_SYNC_SOURCE, SMITHERS_ENGINE_WORKTREE_SYNC_PATCH, "task worktree sync"], + [ + SMITHERS_ENGINE_WORKTREE_CREATE_FETCH_SOURCE, + SMITHERS_ENGINE_WORKTREE_CREATE_FETCH_PATCH, + "task worktree creation fetch" + ], [ SMITHERS_ENGINE_AGENT_USAGE_PROGRESS_SOURCE, SMITHERS_ENGINE_AGENT_USAGE_PROGRESS_PATCH, @@ -7914,7 +8077,10 @@ function compileTask(input: { timeoutMs, heartbeatTimeoutMs, retries, - retryPolicy: { backoff: "exponential", initialDelayMs: 1_000 }, + // Retries wait 60s, 120s, 240s, then Smithers' 5-minute cap. A 1s base + // spent a three-attempt budget in about 25s, inside the minute a + // contended Claude Code OAuth refresh can take to clear (#1084). + retryPolicy: { backoff: "exponential", initialDelayMs: 60_000 }, ...(input.sourceRevision === undefined ? {} : { sourceRevision: input.sourceRevision, sourceRef: input.sourceRef! }), @@ -8505,6 +8671,12 @@ function renderWorkflowSource(compiled: CompiledSmithersWorkflow, config: Resolv }); } +/** + * Rebind each declared output's schema to the bundle this build installs into task workspaces (#982). + * The contract digest and validator build keep their recorded values: they only record the build + * that planned the output, and the refreshed workflow copies them into its markers (and, for a + * dynamic run, its runtime task plan), which are compared with the run's sealed plan (#921). + */ function taskWithCurrentArtifactSchemas(task: CompiledSmithersTask): CompiledSmithersTask { return { ...task, @@ -8517,7 +8689,7 @@ function taskWithCurrentArtifactSchemas(task: CompiledSmithersTask): CompiledSmi return { path: output.path, contract: output.contract, - contractDigest: artifactContractDefinition(output.contract).digest, + contractDigest: output.contractDigest, primary: output.primary, ...(binding === undefined ? {} @@ -8526,7 +8698,7 @@ function taskWithCurrentArtifactSchemas(task: CompiledSmithersTask): CompiledSmi schemaId: binding.schema_id, schemaSha256: binding.schema_sha256, schemaBundleSha256: binding.schema_bundle_sha256, - validatorBuild: binding.validator_build + validatorBuild: output.validatorBuild ?? binding.validator_build }) }; }) diff --git a/packages/runtime/src/start-run.ts b/packages/runtime/src/start-run.ts index ba32e842b..7c2eb10d7 100644 --- a/packages/runtime/src/start-run.ts +++ b/packages/runtime/src/start-run.ts @@ -17,6 +17,7 @@ import { readRegularFileSnapshot, readRunMetadataDocument, readRunState, + replayEvents, safeResolveInside, sensitiveEnvironmentValues, updateRunStatus, @@ -54,6 +55,7 @@ import { prepareTrustedCliEnvironment, runTrustedJsonValidatorPreflight, TRUSTED_CLI_ENVIRONMENT_VARIABLES, + ULTRAFUZZ_TRUSTED_BIN_ENV, type TrustedCliEnvironment } from "./trusted-cli.js"; import { hasRuntimeErrors, runtimeFailure, runtimeResult } from "./utils.js"; @@ -189,12 +191,19 @@ function controllerRefreshInspectionEnvironment( } export async function startRun(input: StartRunInput) { + let createdLayout: RunLayout | undefined; const planned = await planRun(input, { enforceDataGovernance: true, beforeMaterialize: async ({ resolvedConfig, expandedGraph }) => - requiredCommandPreflightDiagnostics(input, resolvedConfig, expandedGraph) + requiredCommandPreflightDiagnostics(input, resolvedConfig, expandedGraph), + afterLayoutCreated: (layout) => { + createdLayout = layout; + } }); if (!planned.ok || !planned.value) { + if (createdLayout !== undefined) { + recordLaunchFailure(createdLayout, planned.diagnostics, sensitiveEnvironmentValues(input.env ?? process.env)); + } return runtimeFailure(planned.diagnostics); } @@ -209,7 +218,9 @@ export async function startRun(input: StartRunInput) { try { releaseControlLock = await acquireWorkflowControlLock(plan.layout); } catch (error) { - return runtimeFailure([smithersDiagnostic(error, "WORKFLOW_CONTROL_PREPARATION_FAILED")]); + const diagnostic = smithersDiagnostic(error, "WORKFLOW_CONTROL_PREPARATION_FAILED"); + recordLaunchFailure(plan.layout, [diagnostic], forbiddenSecretValues); + return runtimeFailure([diagnostic]); } try { const compiled = compileSmithersWorkflow({ @@ -331,13 +342,7 @@ export async function startRun(input: StartRunInput) { ); } catch (error) { const diagnostic = smithersDiagnostic(error, "WORKFLOW_SUBMISSION_FAILED"); - updateRunStatus(plan.layout, "failed", undefined, { forbiddenSecretValues }); - appendEvent(plan.layout, { - eventType: "workflow-submit-failed", - status: "failed", - payload: workflowSubmissionFailureEventPayload(diagnostic), - forbiddenSecretValues - }); + recordLaunchFailure(plan.layout, [diagnostic], forbiddenSecretValues); return runtimeFailure([diagnostic]); } finally { await releaseControlLock(); @@ -478,6 +483,7 @@ export async function resumeRun(input: WorkflowLifecycleInput) { async function submitSmithersContinuation(input: WorkflowLifecycleInput) { let releaseLifecycleLock: (() => Promise) | undefined; + const diagnostics: RuntimeDiagnostic[] = []; try { const projectRoot = path.resolve(input.projectRoot); const runsRoot = await runsRootForProject(projectRoot); @@ -485,6 +491,11 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { const layout = layoutForRunRoot(path.join(runsRoot, runId), runId); assertPathInside(runsRoot, layout.root, "run root"); if (fs.existsSync(runsRoot)) assertNoSymlinkComponents(runsRoot, layout.root, "run root"); + // Checked before the lifecycle lock, whose directory a launch that failed before compiling never created. + const launchFailure = pathIsMissing(workflowControlPaths(projectRoot, layout).integrityPath) + ? recordedLaunchFailure(layout) + : undefined; + if (launchFailure !== undefined) return runtimeFailure([launchFailure]); releaseLifecycleLock = await acquireWorkflowLifecycleLock(layout); assertRegularFileInside(layout.root, layout.runMetadataPath, "run metadata"); @@ -526,6 +537,10 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { const smithersRoot = safeResolveInside(layout.root, "smithers", "Smithers evidence"); const tasksPath = safeResolveInside(smithersRoot, "tasks.json", "workflow task manifest"); const configPath = safeResolveInside(smithersRoot, "resolved-config.json", "workflow config"); + // Agent adapters parse ULTRAFUZZ_CONFIG_PATH as TOML; given the JSON above + // they find no agent tables and fall back to default auth. Launch writes + // the same config as TOML beside it and hands adapters a copy of that file. + const agentConfigPath = safeResolveInside(smithersRoot, "execution-config.toml", "workflow agent config"); let taskDocument: SmithersTaskManifestDocument | undefined; let config: ResolvedConfig | undefined; if (fs.existsSync(tasksPath)) { @@ -571,7 +586,7 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { ...forgeGuard.env, ULTRAFUZZ_ARTIFACTS_MODULE: import.meta.resolve("@ultrafuzz/artifacts"), ULTRAFUZZ_RUNTIME_MODULE: import.meta.resolve("@ultrafuzz/runtime"), - ...(config === undefined ? {} : { ULTRAFUZZ_CONFIG_PATH: configPath }), + ...(config === undefined ? {} : { ULTRAFUZZ_CONFIG_PATH: agentConfigPath }), ULTRAFUZZ_WORKFLOW_PERSISTED_PATH: workflowPath }; let trustedCli: TrustedCliEnvironment = { @@ -593,9 +608,27 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { }); if (prepared.active) runTrustedJsonValidatorPreflight({ layout, trusted: prepared }); trustedCli = prepared; - } catch { + } catch (error) { // Historical validator identity is task setup provenance, not authority - // to prevent Smithers from continuing the workflow. + // to prevent Smithers from continuing the workflow. Keep the run-owned + // launcher first on PATH anyway: it re-verifies its closure on every + // call, while dropping it lets tasks run whatever `ultrafuzz` is on PATH. + const launcher = path.join( + layout.root, + "trusted-bin", + process.platform === "win32" ? "ultrafuzz.cmd" : "ultrafuzz" + ); + const launcherKept = fs.existsSync(launcher); + if (launcherKept) trustedCli.env[ULTRAFUZZ_TRUSTED_BIN_ENV] = path.dirname(launcher); + diagnostics.push( + resumeWarning( + "WORKFLOW_TRUSTED_CLI_UNVERIFIED", + launcherKept + ? `resume could not re-verify the run's trusted Ultrafuzz CLI (tasks still call ${launcher}, and their preflight-json-validator step fails while that launcher cannot verify itself)` + : `resume could not re-verify the run's trusted Ultrafuzz CLI (${launcher} does not exist, so tasks call whatever \`ultrafuzz\` is on PATH)`, + error + ) + ); } } const agentRefs = tasks.flatMap((task) => task.agentChain.map((profile) => profile.agentRef)); @@ -611,7 +644,15 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { assertCurrentCloudAgentCredentialEnvironment(config, tasks, lifecycleEnvironment); } if (typeof metadata.source_revision === "string") { - repairPrunableRunWorktreeRegistrations({ projectRoot, runRoot: layout.root, runId }); + try { + repairPrunableRunWorktreeRegistrations({ projectRoot, runRoot: layout.root, runId }); + } catch (error) { + // Pruning is cleanup: a stale registration it leaves behind surfaces + // when Smithers recreates that task's worktree, so do not stop here. + diagnostics.push( + resumeWarning("WORKFLOW_WORKTREE_REPAIR_FAILED", "resume could not prune stale task worktrees", error) + ); + } } const result = await runSmithersLifecycleCommand({ action: "resume", @@ -623,22 +664,6 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { force: input.force, retryFailed: input.retryFailed, priorInspection: refreshInspection, - beforeStoppedReset: async (inspection) => { - if (!hasRetainedResetControlAuthority(projectRoot, layout, workflow)) return; - // Authenticated original v2 plans predate governance and the current - // linked-evidence contract; retain their native legacy reset behavior. - if (authenticatedContinuationGovernancePath(projectRoot, layout, workflow.control_generation) === undefined) - return; - // workflow-sync reads linked evidence through this module. Load the - // failure-only checkpoint after initialization, at an actual reset. - const { preserveFailedWorkflowAttemptsBeforeReset } = await import("./workflow-sync.js"); - await preserveFailedWorkflowAttemptsBeforeReset({ - projectRoot, - runId, - env: lifecycleEnvironment, - inspection - }); - }, relaunchPaths: { runRoot: layout.root, logsDir: path.join(smithersRoot, "logs") @@ -668,40 +693,32 @@ async function submitSmithersContinuation(input: WorkflowLifecycleInput) { trustedCli.environmentVariableNames ) }); - recordNativeContinuationState({ - layout, - config, - requestedConcurrency: input.maxConcurrency, - alreadyRunning: result.alreadyRunning ?? false - }); - return runtimeResult(true, { - run_id: runId, - workflow_run_id: smithersRunId, - action: "resume" as const, - submitted: !result.alreadyRunning - }); + // An attach to a run Smithers still reports active started no controller, + // so it must not re-record status, lease or deadline; the resume that + // starts the next controller does. + if (result.alreadyRunning !== true) { + recordNativeContinuationState({ layout, config, requestedConcurrency: input.maxConcurrency }); + } + return runtimeResult( + true, + { + run_id: runId, + workflow_run_id: smithersRunId, + action: "resume" as const, + submitted: !result.alreadyRunning + }, + diagnostics + ); } catch (error) { - return runtimeFailure([smithersDiagnostic(error, "WORKFLOW_LIFECYCLE_FAILED")]); + return runtimeFailure([ + smithersDiagnostic(error, "WORKFLOW_LIFECYCLE_FAILED"), + ...diagnostics + ]); } finally { await releaseLifecycleLock?.(); } } -function hasRetainedResetControlAuthority( - projectRoot: string, - layout: RunLayout, - workflow: Record -): boolean { - if (Object.hasOwn(workflow, "control_generation") || Object.hasOwn(workflow, "control_integrity_path")) return true; - try { - fs.lstatSync(workflowControlPaths(projectRoot, layout).integrityPath); - return true; - } catch (error) { - if (error instanceof Error && "code" in error && error.code === "ENOENT") return false; - throw error; - } -} - function parseContinuationResolvedConfigBytes(bytes: Uint8Array): ResolvedConfig { try { return parseResolvedConfigJsonBytes(bytes); @@ -739,7 +756,6 @@ function recordNativeContinuationState(input: { layout: RunLayout; config: ResolvedConfig | undefined; requestedConcurrency: number | undefined; - alreadyRunning: boolean; }): void { try { const submittedAt = new Date().toISOString(); @@ -749,19 +765,17 @@ function recordNativeContinuationState(input: { state.status = "running"; state.started_at ??= submittedAt; delete state.finished_at; - if (!input.alreadyRunning) { - const leaseDurationMs = - (input.config?.run.controllerLeaseSeconds ?? Math.max(1, state.controller_lease.duration_ms / 1_000)) * 1_000; - state.controller_lease = { - ...state.controller_lease, - status: "active", - duration_ms: leaseDurationMs, - renewed_at: submittedAt, - expires_at: new Date(submittedAtMs + leaseDurationMs).toISOString() - }; - state.concurrency.requested_concurrency = - input.requestedConcurrency ?? input.config?.run.maxParallelAgents ?? state.concurrency.requested_concurrency; - } + const leaseDurationMs = + (input.config?.run.controllerLeaseSeconds ?? Math.max(1, state.controller_lease.duration_ms / 1_000)) * 1_000; + state.controller_lease = { + ...state.controller_lease, + status: "active", + duration_ms: leaseDurationMs, + renewed_at: submittedAt, + expires_at: new Date(submittedAtMs + leaseDurationMs).toISOString() + }; + state.concurrency.requested_concurrency = + input.requestedConcurrency ?? input.config?.run.maxParallelAgents ?? state.concurrency.requested_concurrency; if (input.config !== undefined) { state.workflow_deadline_at = new Date( submittedAtMs + input.config.run.workflowDeadlineSeconds * 1_000 @@ -779,6 +793,11 @@ function recordNativeContinuationState(input: { } } +function resumeWarning(code: string, context: string, error: unknown): RuntimeDiagnostic { + const diagnostic = smithersDiagnostic(error, code); + return { ...diagnostic, message: `${context}: ${diagnostic.message}`, severity: "warning", source: "runtime" }; +} + export async function replayRun(input: WorkflowLifecycleInput) { return submitLifecycleAction(input, "replay"); } @@ -789,7 +808,12 @@ export async function forkRun(input: WorkflowLifecycleInput) { export async function pauseRun(input: PauseRunInput) { const projectRoot = path.resolve(input.projectRoot); - const evidence = await readLinkedWorkflowEvidence(projectRoot, input.runId); + // Like `cancel`, pausing only asks the runner to park the linked run, so diverged control + // documents must not block it. + const evidence = await readLinkedWorkflowEvidence(projectRoot, input.runId, { + tolerateControlDivergence: true, + observeOnly: true + }); if (!evidence.ok) { return runtimeFailure(evidence.diagnostics); } @@ -833,13 +857,53 @@ export async function pauseRun(input: PauseRunInput) { } } -function workflowSubmissionFailureEventPayload( - diagnostic: RuntimeDiagnostic -): Extract["payload"] { - if (!isWorkflowSubmissionFailureEventPayload(diagnostic)) { - throw new Error("workflow submission diagnostic does not match the current event contract"); +/** + * A launch that fails after its run directory exists must not look pending, running, or resumable: + * mark the run failed and keep the original error, which readers report in place of a missing seal. + * The event contract allows one code, so any other code is kept in the message. + */ +function recordLaunchFailure( + layout: RunLayout, + diagnostics: readonly RuntimeDiagnostic[], + forbiddenSecretValues: readonly string[] +): void { + const errors = diagnostics.filter((diagnostic) => diagnostic.severity === "error"); + const [only] = errors; + const payload = + errors.length === 1 && only !== undefined && isWorkflowSubmissionFailureEventPayload(only) + ? only + : { + code: "WORKFLOW_SUBMISSION_FAILED" as const, + message: errors.map((diagnostic) => `${diagnostic.code}: ${diagnostic.message}`).join("; "), + severity: "error" as const, + source: "workflow" as const, + details: {} + }; + try { + updateRunStatus(layout, "failed", undefined, { forbiddenSecretValues }); + appendEvent(layout, { eventType: "workflow-submit-failed", status: "failed", payload, forbiddenSecretValues }); + } catch { + // Best effort: the caller returns the original diagnostics either way. + } +} + +/** A failed launch's recorded error; callers ask only about runs whose workflow controls were never sealed. */ +function recordedLaunchFailure(layout: RunLayout): RuntimeDiagnostic | undefined { + try { + const failure = replayEvents(layout, Number.MAX_SAFE_INTEGER) + .records.filter((record) => record.event_type === "workflow-submit-failed") + .at(-1); + if (failure?.event_type !== "workflow-submit-failed") return undefined; + return { + code: "RUN_LAUNCH_FAILED", + message: `run ${layout.runId} failed to launch: ${failure.payload.message}. A run whose launch failed cannot be resumed; fix the cause and start a new run`, + severity: "error", + source: "workflow", + path: layout.eventsPath + }; + } catch { + return undefined; } - return diagnostic; } function isWorkflowSubmissionFailureEventPayload( @@ -1094,23 +1158,14 @@ function parseSealedTaskManifestForObserver(contents: Readonly<{ graph: Buffer; /** * `assertSealedPlannedGraph` does two different jobs behind one name. The first is structural: the * bytes must validate against the planned-graph schema, and nothing can report on a document that is - * not a planned graph at all. The second, `assertPlannedGraphSemantics`, re-derives the graph against - * *this build* — it looks every output contract up in the running process's artifact-contract registry - * and insists the digests, schema IDs and validator build recorded at compile time still match what - * this checkout produces. - * - * That second job is not a property of the run; it is a property of the tree observing the run. An - * operator whose checkout has moved on since the run was submitted -- a rebased branch, a newer - * release, a contract whose schema was revised -- gets `planned graph output schema binding changed` - * and loses `status` for a run that is otherwise intact and possibly still executing. That is the same - * failure as issue #866, one throw further along the same read-only path: the run is fine, the - * observer's registry disagrees, and the operator is the one punished. `packages/artifacts`'s own - * semantic-gate collects exactly this condition as an issue rather than raising it, so the softer - * reading already exists in the codebase. + * not a planned graph at all. The second, `assertPlannedGraphSemantics`, checks the graph's internal + * consistency (unique IDs, dependency joins, artifact directories, loop and model coordinates) with + * the rules of the build doing the reading. * - * Execution must still refuse: running a node whose output contract no longer matches the registry - * that will validate its artifacts would produce evidence nothing can check. So the downgrade is - * observer-only, and schema invalidity stays fatal for everyone. + * Those rules belong to the tree observing the run, not to the run: a sealed graph that a newer + * checkout's rules reject is still the graph the run is executing, and losing `status` for it is the + * same failure as issue #866, one throw further along the same read-only path. So the downgrade is + * observer-only: execution callers still refuse, and schema invalidity stays fatal for everyone. */ function parseSealedPlannedGraphForObserver(graphBytes: Buffer): { graph: PlannedGraphDocument; @@ -1121,12 +1176,12 @@ function parseSealedPlannedGraphForObserver(graphBytes: Buffer): { if (!validatePlannedGraph(value).ok) return { graph: assertSealedPlannedGraph(value), divergences: [] }; const graph = value as PlannedGraphDocument; try { - assertPlannedGraphSemantics(graph, { allowHistoricalSchemaBundle: true }); + assertPlannedGraphSemantics(graph); } catch (error) { return { graph, divergences: [ - `sealed planned graph no longer re-derives against this build's artifact contracts: ${error instanceof Error ? error.message : String(error)}` + `sealed planned graph fails this build's planned-graph semantics: ${error instanceof Error ? error.message : String(error)}` ] }; } @@ -1323,8 +1378,9 @@ export async function readLinkedWorkflowEvidence( throw new Error("run metadata workflow IDs do not exactly match the active workflow run"); } - // Observers pass `tolerateControlDivergence` so a divergent control file downgrades to a reported - // warning instead of hiding a live run entirely (issue #674). Execution callers omit it and keep + // Observers, `pause` and `cancel` pass `tolerateControlDivergence` so a divergent control file is + // collected in `divergences` instead of failing the read: observers report it as a warning, and + // `pause` and `cancel` still reach the runner (issue #674). Execution callers omit it and keep // failing closed. const tolerateDivergence = options.tolerateControlDivergence === true; const verifiedControl = verifyWorkflowControlSnapshot(resolvedProjectRoot, layout, { tolerateDivergence }); @@ -1471,7 +1527,7 @@ function invalidLinkedWorkflowEvidence(metadataPath: string, error: unknown): Li /** * A pending state records incomplete launch preparation, not launcher liveness. * It also survives an interrupted launch. Unreadable state retains the strict - * missing-seal diagnostic. + * missing-seal or missing-journal diagnostic. */ function runStatusIsPreSubmission(layout: RunLayout): boolean { try { @@ -1496,6 +1552,8 @@ function missingLinkedWorkflowEvidenceDiagnostic( path: controlSealPath }; } + const launchFailure = recordedLaunchFailure(layout); + if (launchFailure !== undefined) return launchFailure; return { code: "WORKFLOW_CONTROL_SEAL_MISSING", message: `run ${layout.runId} lacks the required workflow control seal; it may predate sealed runs or be incomplete and cannot be safely upgraded in place. Preserve its stored artifacts and start a new run with a new run ID`, @@ -1506,6 +1564,18 @@ function missingLinkedWorkflowEvidenceDiagnostic( } const linkJournalPath = workflowRunLinkJournalPath(layout); if (pathIsMissing(linkJournalPath)) { + // Launch writes this journal only after publishing the execution snapshot, its longest step, so a + // lock-free reader such as `status`, `pause` or `cancel` can meet a launch still in progress here. + // Report it with the seal's pending code, which `status` already renders as an incomplete launch. + if (runStatusIsPreSubmission(layout)) { + return { + code: "WORKFLOW_CONTROL_SEAL_PENDING", + message: `run ${layout.runId} has incomplete launch preparation: its workflow-link journal has not been written. Launcher liveness is unknown. If the original launch is still active, wait for it to finish; otherwise inspect its error before retrying`, + severity: "warning", + source: "workflow", + path: linkJournalPath + }; + } return { code: "WORKFLOW_RUN_LINK_JOURNAL_MISSING", message: `run ${layout.runId} lacks the required authenticated workflow-link journal; it may predate authenticated lifecycle links or be incomplete and cannot be safely upgraded in place. Preserve its stored artifacts and start a new run with a new run ID`, diff --git a/packages/runtime/src/state-export.ts b/packages/runtime/src/state-export.ts index 157b8564c..55a1579fd 100644 --- a/packages/runtime/src/state-export.ts +++ b/packages/runtime/src/state-export.ts @@ -44,8 +44,13 @@ import { summarizeRunProgress } from "./run-progress.js"; import { diagnosticFromError, runtimeFailure, runtimeResult } from "./utils.js"; import { parseCurrentSmithersInspect, runSmithersInspectionCommand, type SmithersCommandSnapshot } from "./smithers.js"; import { workflowControlDivergenceDiagnostics } from "./control-divergence-diagnostics.js"; -import { linkedWorkflowExecutionEnvironment, readLinkedWorkflowEvidence } from "./start-run.js"; import { + linkedWorkflowExecutionEnvironment, + readLinkedWorkflowEvidence, + type LinkedWorkflowEvidence +} from "./start-run.js"; +import { + TRANSIENT_SYNC_DIAGNOSTIC_CODES, describeObservationSynchronizationDeadline, observationSynchronizationDeadline, synchronizeLinkedWorkflowRun @@ -269,10 +274,7 @@ export async function getRunHealth(input: { const syncDiagnostics: RuntimeDiagnostic[] = [...controlDiagnostics]; if (controlDiagnostics.length === 0) { syncDiagnostics.push( - ...(await synchronizeObservedWorkflowRun( - { projectRoot, runId: input.runId, env: input.env }, - evidence.layout.root - )) + ...(await synchronizeObservedWorkflowRun({ projectRoot, runId: input.runId, env: input.env }, evidence)) ); } else { syncDiagnostics.push({ @@ -307,7 +309,7 @@ export async function getRunHealth(input: { })) ]); } - const health = parseRunHealth(snapshot.json, evidence.smithersRunId); + const health = parseRunHealth(snapshot.json, evidence.smithersRunId, input.runId); if (health === undefined) { return runtimeFailure([ ...syncDiagnostics, @@ -358,35 +360,46 @@ export async function getRunHealth(input: { * Observe-only synchronization reads state.json and run.json strictly while the run's controller, or * another concurrent `status`, keeps replacing them by atomic rename, so it can hit the same transient * snapshot race the direct reads in `getRunHealth` retry. It is retried within the same bounded - * budget. Once that budget is spent the run is still reported: health comes from the workflow runner - * and local run state is simply the last coherent snapshot, which is said in a warning, the way a - * skipped synchronization is reported. Every other failure propagates unchanged. + * budget, and a retry reads the run's evidence again rather than reusing the caller's. * - * The refresh is also bounded by the opt-in observation deadline when one is configured. Health below - * comes from the direct runner query, so an exceeded deadline is a warning about possibly stale local - * state, not a failure. One absolute deadline spans every retry, so racing reads cannot extend the - * observer's wall-clock budget. + * Health below comes from the direct runner query, so a refresh never hides it. A thrown refresh + * error, an exhausted race budget, and the transient codes `stats` also tolerates (a failed or + * malformed runner query, an exceeded opt-in observation deadline, a lock the pass could not take) + * become warnings that local run state may be stale, so they neither fail `status` nor stop + * `--watch`. Any other error the refresh returns keeps its severity. One absolute deadline spans + * every retry, so racing reads cannot extend the observer's wall-clock budget. */ -async function synchronizeObservedWorkflowRun(input: SyncRunInput, runRoot: string): Promise { +async function synchronizeObservedWorkflowRun( + input: SyncRunInput, + evidence: LinkedWorkflowEvidence +): Promise { const deadlineMs = observationSynchronizationDeadline(input.env); + let firstAttempt = true; try { - const sync = await retryTransientSnapshotObservation(() => - synchronizeLinkedWorkflowRun(input, { observeOnly: true, deadlineMs }) - ); + const sync = await retryTransientSnapshotObservation(() => { + const reuseEvidence = firstAttempt; + firstAttempt = false; + return synchronizeLinkedWorkflowRun(input, { + observeOnly: true, + tolerateInvalidEventStreams: true, + deadlineMs, + ...(reuseEvidence ? { evidence } : {}) + }); + }); return sync.diagnostics.map((diagnostic) => - diagnostic.code === "WORKFLOW_SYNC_DEADLINE_EXCEEDED" + TRANSIENT_SYNC_DIAGNOSTIC_CODES.has(diagnostic.code) ? describeObservationSynchronizationDeadline({ ...diagnostic, severity: "warning" as const }) : diagnostic ); } catch (error) { - if (!isTransientSnapshotRace(error)) throw error; + const reason = error instanceof Error ? error.message : String(error); return [ { - code: "WORKFLOW_STATE_SYNC_RACED", - message: `run state synchronization was skipped because ${error.message}; reported counts come from the workflow runner and local run state may be stale`, + code: isTransientSnapshotRace(error) ? "WORKFLOW_STATE_SYNC_RACED" : "WORKFLOW_STATE_SYNC_FAILED", + message: `run state synchronization was skipped because ${reason}; reported counts come from the workflow runner and local run state may be stale`, severity: "warning", source: "runtime", - path: runRoot + path: evidence.layout.root } ]; } @@ -830,7 +843,8 @@ export const CURRENT_SMITHERS_STATUS_KEY_CONTRACT = { function parseRunHealth( value: unknown, - expectedWorkflowRunId: string + expectedWorkflowRunId: string, + runId: string ): | Omit | undefined { @@ -978,7 +992,7 @@ function parseRunHealth( return { workflow_status: workflowStatus, verdict, - reason: publicHealthReason(reason), + reason: publicHealthReason(reason, runId), counts: typedCounts, model_mix: modelMix, throughput: { @@ -1165,9 +1179,15 @@ function parseRunHealthOneshotControl(value: unknown): RunHealthValue["oneshot_c }; } -function publicHealthReason(value: string): string { - // `ultrafuzz why` now wraps the engine diagnosis, so recommend it directly. - return value.replace(/`?smithers\s+why`?/giu, "`ultrafuzz why`").replace(/smithers/giu, "workflow runner"); +function publicHealthReason(value: string, runId: string): string { + // Point at the Ultrafuzz commands that wrap the runner's own: `ultrafuzz why` for its diagnosis and + // `ultrafuzz resume` to continue an orphaned run. The run ID is joined in after the generic rename, + // which would otherwise rewrite an ID that contains "smithers". + return value + .replace(/`?smithers\s+why`?/giu, "`ultrafuzz why`") + .split(/`?smithers\s+supervise\s+-r\s+[^\s`;,]+`?/giu) + .map((part) => part.replace(/smithers/giu, "workflow runner")) + .join(`\`ultrafuzz resume ${runId}\``); } function objectRecord(value: unknown): Record | undefined { diff --git a/packages/runtime/src/templates/smithers/agents/deepseek.tsx b/packages/runtime/src/templates/smithers/agents/deepseek.tsx index 3e85d0296..4fe9c08b6 100644 --- a/packages/runtime/src/templates/smithers/agents/deepseek.tsx +++ b/packages/runtime/src/templates/smithers/agents/deepseek.tsx @@ -1,8 +1,8 @@ import { readFileSync } from "node:fs"; +import path from "node:path"; import { ClaudeCodeAgent as SmithersClaudeCodeAgent } from "smthrs"; import { workflowControlChildEnvironment, workflowControlCredentialValue } from "./environment"; import { resolveProviderHome } from "./provider-home"; -import { parseStrictJson } from "./strict-json"; import { readStringTable, stringField } from "./toml"; type DeepSeekAuthConfig = { auth?: string; api_key_env?: string; config_dir?: string }; @@ -11,45 +11,18 @@ export type DeepSeekTaskOptions = { model?: string; reasoningEffort?: string; ad type DeepSeekAgentOptions = ConstructorParameters[0] & DeepSeekAuthOptions; type DeepSeekCommandParams = Parameters[0]; type DeepSeekCommand = Awaited>; -type DeepSeekOutputInterpreter = ReturnType; type DeepSeekReasoningEffort = "low" | "high" | "max"; -type DeepSeekUsage = { - inputTokens: number; - outputTokens: number; - cacheReadTokens: number; - cacheWriteTokens: 0; - totalTokens: number; -}; -type DeepSeekSmithersUsage = { - inputTokens: number; - inputTokenDetails: { noCacheTokens: number; cacheReadTokens: number; cacheWriteTokens: 0 }; - outputTokens: number; - outputTokenDetails: { textTokens: undefined; reasoningTokens: undefined }; - totalTokens: number; -}; const DEEPSEEK_ANTHROPIC_BASE_URL = "https://api.deepseek.com/anthropic"; const DEEPSEEK_REASONING_EFFORTS = ["low", "high", "max"] as const; -const DEEPSEEK_RESULT_MAX_BYTES = 1024 * 1024; -const DEEPSEEK_RESULT_MAX_DEPTH = 32; -const DEEPSEEK_RESULT_MAX_ITEMS = 10_000; -const DEEPSEEK_RESULT_MAX_PROPERTIES = 10_000; -const DEEPSEEK_LEGACY_USAGE_FIELDS = [ - "input_tokens", - "inputTokens", - "outputTokens", - "completion_tokens", - "cache_read_input_tokens", - "cacheReadTokens", - "cached_input_tokens" -] as const; /** * Runs the Claude Code harness against DeepSeek's Anthropic-compatible * endpoint: a compatibility pairing, not DeepSeek's first-party coding agent. - * Keep it as a distinct factory so credentials, model selection, telemetry - * semantics, and pricing provenance never inherit Anthropic defaults - * accidentally. + * Keep it as a distinct factory so credentials, model selection, and pricing + * provenance never inherit Anthropic defaults accidentally. Token usage needs + * no adapter code: Claude Code reports it under Anthropic field names, which + * Smithers' ClaudeCodeAgent already reads. */ export function createDeepSeekAgent(options: DeepSeekTaskOptions = {}): SmithersClaudeCodeAgent { const reasoningEffort = deepSeekReasoningEffort(options.reasoningEffort); @@ -66,7 +39,6 @@ export function createDeepSeekAgent(options: DeepSeekTaskOptions = {}): Smithers } export class DeepSeekClaudeCodeAgent extends SmithersClaudeCodeAgent { - private pendingUsage: DeepSeekSmithersUsage | undefined; private readonly ultrafuzzApiKey: string; // Smithers 0.35.0's BaseCliAgent rejects unknown constructor options with a @@ -78,35 +50,7 @@ export class DeepSeekClaudeCodeAgent extends SmithersClaudeCodeAgent { this.ultrafuzzApiKey = ultrafuzzApiKey; } - override generate( - ...args: Parameters - ): ReturnType { - return this.withDeepSeekUsage(super.generate(...args)) as ReturnType; - } - - override stream( - ...args: Parameters - ): ReturnType { - return this.withDeepSeekStreamUsage(super.stream(...args)) as ReturnType; - } - - override createOutputInterpreter(): DeepSeekOutputInterpreter { - const base = super.createOutputInterpreter(); - return { - ...base, - onStdoutLine: (line) => { - const usage = deepSeekUsageFromResultLine(line); - if (usage !== undefined) this.pendingUsage = deepSeekSmithersUsage(usage); - const events = base.onStdoutLine?.(line) ?? []; - if (usage === undefined) return events; - const completedUsage = deepSeekCompletedUsage(usage); - return events.map((event) => (event.type === "completed" ? { ...event, usage: completedUsage } : event)); - } - }; - } - override async buildCommand(params: DeepSeekCommandParams): Promise { - this.pendingUsage = undefined; this.opts.settingSources = ""; const command = await super.buildCommand(params); try { @@ -160,28 +104,6 @@ export class DeepSeekClaudeCodeAgent extends SmithersClaudeCodeAgent { throw error; } } - - private withDeepSeekUsage(promise: Promise): Promise { - return promise - .then((result) => attachDeepSeekResultUsage(result, this.pendingUsage)) - .catch((error: unknown) => { - throw attachDeepSeekFailureUsage(error, this.pendingUsage); - }) - .finally(() => { - this.pendingUsage = undefined; - }); - } - - private withDeepSeekStreamUsage(promise: Promise): Promise { - return promise - .then((result) => attachDeepSeekStreamUsage(result, this.pendingUsage)) - .catch((error: unknown) => { - throw attachDeepSeekFailureUsage(error, this.pendingUsage); - }) - .finally(() => { - this.pendingUsage = undefined; - }); - } } function deepSeekAuthOptions(): DeepSeekAuthOptions { @@ -221,136 +143,3 @@ function deepSeekReasoningEffort(value: string | undefined): DeepSeekReasoningEf } throw new Error(`DeepSeekAgent reasoning effort must be one of ${DEEPSEEK_REASONING_EFFORTS.join(", ")}: ${value}`); } - -/** - * DeepSeek bills cache misses and cache hits independently. Its completion - * token count already includes thinking tokens, so exposing a separate - * reasoning count would double-count both tokens and spend. - */ -function deepSeekUsageFromResultLine(line: string): DeepSeekUsage | undefined { - const first = firstNonJsonWhitespace(line); - if (first === undefined) return undefined; - const objectCandidate = first === "{"; - let payload: unknown; - try { - payload = parseStrictJson(line, { - maxBytes: DEEPSEEK_RESULT_MAX_BYTES, - maxDepth: DEEPSEEK_RESULT_MAX_DEPTH, - maxItems: DEEPSEEK_RESULT_MAX_ITEMS, - maxProperties: DEEPSEEK_RESULT_MAX_PROPERTIES - }); - } catch (error) { - if (objectCandidate) { - throw new Error( - `DeepSeek result output is invalid strict JSON: ${error instanceof Error ? error.message : String(error)}`, - { cause: error } - ); - } - return undefined; - } - if (!isRecord(payload) || payload.type !== "result") return undefined; - if (!isRecord(payload.usage)) throw new Error("DeepSeek result usage must be an object"); - const usage = payload.usage; - for (const legacyField of DEEPSEEK_LEGACY_USAGE_FIELDS) { - if (Object.prototype.hasOwnProperty.call(usage, legacyField)) { - throw new Error(`DeepSeek result usage contains unsupported legacy alias ${legacyField}`); - } - } - const inputTokens = requiredDeepSeekTokenCount(usage, "prompt_cache_miss_tokens"); - const outputTokens = requiredDeepSeekTokenCount(usage, "output_tokens"); - const cacheReadTokens = requiredDeepSeekTokenCount(usage, "prompt_cache_hit_tokens"); - const normalized = { - inputTokens, - outputTokens, - cacheReadTokens, - cacheWriteTokens: 0 as const - }; - const totalTokens = normalized.inputTokens + normalized.cacheReadTokens + normalized.outputTokens; - if (!Number.isSafeInteger(totalTokens)) throw new Error("DeepSeek result usage exceeds the safe integer range"); - return { ...normalized, totalTokens }; -} - -function firstNonJsonWhitespace(value: string): string | undefined { - for (const character of value) { - if (character !== " " && character !== "\t" && character !== "\n" && character !== "\r") return character; - } - return undefined; -} - -function requiredDeepSeekTokenCount(value: Record, field: string): number { - const candidate = value[field]; - if (typeof candidate !== "number" || !Number.isSafeInteger(candidate) || candidate < 0) { - throw new Error(`DeepSeek result usage.${field} must be a non-negative safe integer`); - } - return candidate; -} - -function deepSeekCompletedUsage(usage: DeepSeekUsage): Record { - return { - input_tokens: deepSeekProviderInputTokens(usage), - fresh_input_tokens: usage.inputTokens, - output_tokens: usage.outputTokens, - cache_read_input_tokens: usage.cacheReadTokens, - cache_creation_input_tokens: usage.cacheWriteTokens, - total_tokens: usage.totalTokens - }; -} - -function deepSeekSmithersUsage(usage: DeepSeekUsage): DeepSeekSmithersUsage { - return { - inputTokens: deepSeekProviderInputTokens(usage), - inputTokenDetails: { - noCacheTokens: usage.inputTokens, - cacheReadTokens: usage.cacheReadTokens, - cacheWriteTokens: usage.cacheWriteTokens - }, - outputTokens: usage.outputTokens, - outputTokenDetails: { - textTokens: undefined, - reasoningTokens: undefined - }, - totalTokens: usage.totalTokens - }; -} - -function deepSeekProviderInputTokens(usage: DeepSeekUsage): number { - return usage.totalTokens - usage.outputTokens; -} - -function attachDeepSeekResultUsage(result: T, usage: DeepSeekSmithersUsage | undefined): T { - if (usage === undefined || !isRecord(result)) return result; - try { - result.usage = usage; - result.totalUsage = usage; - } catch { - // Telemetry must never turn a successful provider invocation into a model - // failure if an exotic Smithers result becomes immutable. - } - return result; -} - -function attachDeepSeekStreamUsage(result: T, usage: DeepSeekSmithersUsage | undefined): T { - if (usage === undefined || !isRecord(result)) return result; - try { - result.usage = Promise.resolve(usage); - result.totalUsage = Promise.resolve(usage); - } catch { - // Telemetry must never turn a successful provider invocation into a model - // failure if an exotic Smithers stream result becomes immutable. - } - return result; -} - -function attachDeepSeekFailureUsage(error: unknown, usage: DeepSeekSmithersUsage | undefined): unknown { - if (usage === undefined || !isRecord(error)) return error; - try { - error.usage = usage; - } catch { - // Preserve the original failure if an exotic error object is immutable. - } - return error; -} - -function isRecord(value: unknown): value is Record { - return typeof value === "object" && value !== null && !Array.isArray(value); -} diff --git a/packages/runtime/src/templates/smithers/agents/environment.tsx b/packages/runtime/src/templates/smithers/agents/environment.tsx index 1b44d8573..85829b6e2 100644 --- a/packages/runtime/src/templates/smithers/agents/environment.tsx +++ b/packages/runtime/src/templates/smithers/agents/environment.tsx @@ -1,4 +1,3 @@ -import { createHash } from "node:crypto"; import { existsSync } from "node:fs"; import path from "node:path"; import { fileURLToPath } from "node:url"; @@ -7,11 +6,33 @@ import { parseStrictJsonBytes, readRegularFileSnapshot } from "./strict-json"; export const PROVIDER_SCOPED_SENSITIVE_ENVIRONMENT_CAPABILITY = "ultrafuzz.provider-scoped-sensitive-environment.v1" as const; +// The controller computes acknowledged route IDs and credential ownership +// with these @ultrafuzz/runtime functions. Use the same functions, loaded from +// the module the rendered workflow imports, rather than copies that have to be +// kept in step with them. +const { + isCredentialLikeEnvironmentVariableName, + providerRouteDestination, + routeOwnsCredentialLikeEnvironmentVariable +} = (await import( + process.env.ULTRAFUZZ_RUNTIME_MODULE ?? + new URL("../../modules/@ultrafuzz/runtime/dist/index.js", import.meta.url).href +)) as { + isCredentialLikeEnvironmentVariableName: (name: string) => boolean; + providerRouteDestination: ( + agent: string, + env: Record, + routeConfig?: Uint8Array + ) => string; + routeOwnsCredentialLikeEnvironmentVariable: (agent: string, name: string) => boolean; +}; + const CONTROLLER_ONLY_ENVIRONMENT_VARIABLES = [ "SMITHERS_BIN", "SMITHERS_CLI_SRC_DIR", "ULTRAFUZZ_AGENT_ENV_ALLOWLIST", "ULTRAFUZZ_ARTIFACTS_MODULE", + "ULTRAFUZZ_BUN_MODULE_CONFINEMENT", "ULTRAFUZZ_CONFIG_PATH", "ULTRAFUZZ_DATA_DISCLOSURE_ACKNOWLEDGEMENTS", "ULTRAFUZZ_MODAL_PUBLIC_BENCHMARK", @@ -54,26 +75,6 @@ const ENVIRONMENT_VARIABLE_PATTERN = /^[A-Za-z_][A-Za-z0-9_]*$/u; type WorkflowRouteAgent = "ClaudeAgent" | "CodexAgent" | "DeepSeekAgent" | "KimiAgent" | "OpenRouterAgent"; type WorkflowDataRoute = { agent: WorkflowRouteAgent; configDir?: string }; -const ROUTE_ENV_PREFIXES: Readonly> = { - ClaudeAgent: ["ANTHROPIC_", "CLAUDE_CODE_USE_", "AWS_", "AZURE_", "CLOUD_ML_", "FOUNDRY_", "GOOGLE_"], - CodexAgent: ["AZURE_OPENAI_", "OPENAI_"], - KimiAgent: ["KIMI_", "MOONSHOT_"] -}; -const NON_ROUTING_PROVIDER_ENVIRONMENT_NAMES = new Set(["AZURE_EXTENSION_DIR"]); -// Keep this generated, dependency-free boundary in parity with -// @ultrafuzz/security's isSensitiveEnvironmentName contract. -const SENSITIVE_ENVIRONMENT_NAME_PATTERN = - /(?:^|_)(?:API_?KEY|TOKEN|SECRET|PASSWORD|PASSWD|PRIVATE_?KEY|ACCESS_?KEY|CLIENT_?SECRET|CREDENTIALS?|AUTH(?:ORIZATION)?)(?:_|$)/iu; -const ROUTE_PROXY_ENV = [ - "ALL_PROXY", - "HTTP_PROXY", - "HTTPS_PROXY", - "NO_PROXY", - "all_proxy", - "http_proxy", - "https_proxy", - "no_proxy" -] as const; /** * Smithers agents inherit the controller environment by default. Remove every @@ -93,22 +94,12 @@ export function workflowControlChildEnvironment( ].map((name) => [name, ""]) ); const roots = workflowExecutionSnapshotRoots(source); - // Native continuations name the target's workflow, not a sealed execution - // tree. Pi's configured home may live under that target. Continue rejecting - // homes that alias a real advertised snapshot or another controller root. - const piHomeRoots = - source.ULTRAFUZZ_SNAPSHOT_PERSISTED_ROOT || - source.ULTRAFUZZ_SNAPSHOT_PROCESS_ROOT || - source.ULTRAFUZZ_SNAPSHOT_SOURCE_ROOT - ? roots - : workflowExecutionSnapshotRoots({ ...source, ULTRAFUZZ_WORKFLOW_PERSISTED_PATH: undefined }); const aliasesControl = (name: string, value: string): boolean => { // PATH has already passed the controller's command-path admission. Its // trusted-bin/forge guard intentionally lives under the target; blanking // the whole list also removes external CLIs admitted by the controller. if (name === "PATH") return false; - const selectedRoots = name === "PI_CODING_AGENT_DIR" ? piHomeRoots : roots; - return selectedRoots.some((root) => environmentPath(value).includes(root)); + return roots.some((root) => environmentPath(value).includes(root)); }; for (const [name, value] of Object.entries(source)) { if (value !== undefined && aliasesControl(name, value)) child[name] = ""; @@ -164,10 +155,6 @@ function allowlistedEnvironmentVariables( return [...variables.values()]; } -function isCredentialLikeEnvironmentVariableName(name: string): boolean { - return SENSITIVE_ENVIRONMENT_NAME_PATTERN.test(name); -} - function sensitiveAgentEnvironmentVariableNames(source: Record): string[] { const names = new Set(); for (const name of (source.ULTRAFUZZ_SENSITIVE_AGENT_ENV_NAMES ?? "").split(",")) { @@ -198,23 +185,6 @@ function addLogicalEnvironmentVariableNames( } } -function routeOwnsCredentialLikeEnvironmentVariable(agent: WorkflowRouteAgent, name: string): boolean { - const upper = name.toUpperCase(); - let longestPrefix = -1; - const owners = new Set(); - for (const [candidate, prefixes] of Object.entries(ROUTE_ENV_PREFIXES)) { - for (const prefix of prefixes) { - if (!upper.startsWith(prefix) || prefix.length < longestPrefix) continue; - if (prefix.length > longestPrefix) { - longestPrefix = prefix.length; - owners.clear(); - } - owners.add(candidate); - } - } - return owners.has(agent); -} - function restoreRouteScopedAllowlistedCredentials( child: Record, source: Record, @@ -258,127 +228,27 @@ function assertWorkflowDataRoute( required = (record as { required_source_destinations?: unknown }).required_source_destinations; if (!Array.isArray(required) || required.some((entry) => typeof entry !== "string")) throw new Error("sealed data-governance authority is invalid"); - const effective = effectiveWorkflowDataRoute(route, effectiveEnvironment, authorityEnvironment); + const configPath = + route.configDir === undefined || route.agent === "OpenRouterAgent" + ? undefined + : path.join(route.configDir, route.agent === "ClaudeAgent" ? "settings.json" : "config.toml"); + const effective = providerRouteDestination( + route.agent, + { + ...effectiveEnvironment, + ULTRAFUZZ_AGENT_ENV_ALLOWLIST: authorityEnvironment.ULTRAFUZZ_AGENT_ENV_ALLOWLIST, + // The Codex adapter derives OPENAI_BASE_URL from the provider config, + // which the config part of the route already covers. + ...(route.agent === "CodexAgent" && !authorityEnvironment.OPENAI_BASE_URL?.trim() + ? { OPENAI_BASE_URL: undefined } + : {}) + }, + configPath !== undefined && existsSync(configPath) ? readRegularFileSnapshot(configPath, 1024 * 1024) : undefined + ); if (!required.includes(effective)) throw new Error(`effective ${route.agent} provider route changed after disclosure acknowledgement`); } -function effectiveWorkflowDataRoute( - route: WorkflowDataRoute, - source: Record, - authority: Record -): string { - const routeSource = { ...source, ULTRAFUZZ_AGENT_ENV_ALLOWLIST: authority.ULTRAFUZZ_AGENT_ENV_ALLOWLIST }, - routeEnvironment = effectiveRouteEnvironment( - route.agent, - route.agent === "CodexAgent" && !authority.OPENAI_BASE_URL?.trim() - ? { ...routeSource, OPENAI_BASE_URL: undefined } - : routeSource - ); - let configDigest: string | undefined; - if (route.configDir !== undefined && route.agent !== "OpenRouterAgent") { - const configPath = path.join(route.configDir, route.agent === "ClaudeAgent" ? "settings.json" : "config.toml"); - if (existsSync(configPath)) { - const bytes = readRegularFileSnapshot(configPath, 1024 * 1024); - const affectsRoute = - route.agent === "ClaudeAgent" - ? claudeSettingsAffectRoute(bytes) - : route.agent !== "CodexAgent" || codexConfigAffectsRoute(bytes.toString("utf8")); - if (affectsRoute) configDigest = sha256(bytes); - } - } - const digest = - routeEnvironment.length > 0 - ? sha256(JSON.stringify({ agent: route.agent, config: configDigest ?? null, route: routeEnvironment })) - : configDigest, - provider = { - ClaudeAgent: "anthropic", - CodexAgent: "openai", - DeepSeekAgent: "deepseek", - KimiAgent: "moonshot", - OpenRouterAgent: "openrouter" - }[route.agent]; - return digest === undefined - ? `model:${provider}` - : `model:${route.agent.toLowerCase().replace("agent", "")}-route-${digest}`; -} - -/** - * The Codex CLI rewrites its own config.toml on invocation — marketplace - * `last_updated` timestamps, plugin toggles, and project trust levels — so - * digesting the whole file makes the acknowledged route change the moment the - * CLI first runs in a fresh HOME, which failed every sandbox agent task after - * disclosure (#908). Keep this generated copy in exact parity with - * @ultrafuzz/runtime data-governance.ts: only content that can actually - * redirect traffic — a `model_provider` selection, a `[model_providers…]` - * table, or a `base_url` assignment — participates in the route digest. - */ -function codexConfigAffectsRoute(text: string): boolean { - return ( - /(?:^|\n)\s*(?:model_provider|"model_provider"|'model_provider')\s*=/u.test(text) || - /(?:^|\n)\s*\[[^\]\n]*model_providers[^\]\n]*\]/u.test(text) || - /(?:^|\n)\s*(?:base_url|"base_url"|'base_url')\s*=/u.test(text) - ); -} - -function claudeSettingsAffectRoute(bytes: Buffer): boolean { - const parsed = parseStrictJsonBytes(bytes, { - maxBytes: 1024 * 1024, - maxDepth: 32, - maxItems: 4096, - maxProperties: 4096 - }); - if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) - throw new Error("Claude settings must be a JSON object"); - if (Object.keys(parsed).some((name) => /(?:helper|refresh|credentialexport|processwrapper|proxyauth)$/iu.test(name))) - return true; - const configuredEnv = (parsed as Record).env; - if (configuredEnv === undefined) return false; - if (configuredEnv === null || typeof configuredEnv !== "object" || Array.isArray(configuredEnv)) - throw new Error("Claude settings env must be a JSON object"); - return Object.keys(configuredEnv).some((name) => { - const upper = name.toUpperCase(); - return ( - ["ALL_PROXY", "HTTP_PROXY", "HTTPS_PROXY", "NO_PROXY"].includes(upper) || - (!NON_ROUTING_PROVIDER_ENVIRONMENT_NAMES.has(upper) && - !isCredentialLikeEnvironmentVariableName(upper) && - ROUTE_ENV_PREFIXES.ClaudeAgent!.some((prefix) => upper.startsWith(prefix))) - ); - }); -} - -function effectiveRouteEnvironment( - agent: WorkflowRouteAgent, - env: Record -): Array<[string, string]> { - const names = new Set( - (env.ULTRAFUZZ_AGENT_ENV_ALLOWLIST ?? "").split(",").map((entry) => entry.trim().toUpperCase()) - ); - for (const name of ROUTE_PROXY_ENV) names.add(name); - for (const name of Object.keys(env)) - if ( - !isCredentialLikeEnvironmentVariableName(name) && - ROUTE_ENV_PREFIXES[agent]?.some((prefix) => name.startsWith(prefix)) - ) - names.add(name); - if (agent === "CodexAgent") names.add("OPENAI_BASE_URL"); - if (agent === "KimiAgent") names.add("KIMI_BASE_URL"); - for (const name of NON_ROUTING_PROVIDER_ENVIRONMENT_NAMES) names.delete(name); - names.delete("KIMI_CODE_HOME"); - names.delete("KIMI_SHARE_DIR"); - return [...names].sort().flatMap((name): Array<[string, string]> => { - const value = env[name]; - return value !== undefined && - value.trim() !== "" && - !isCredentialLikeEnvironmentVariableName(name) && - (ROUTE_PROXY_ENV.includes(name as never) || ROUTE_ENV_PREFIXES[agent]?.some((prefix) => name.startsWith(prefix))) - ? [[name, value]] - : []; - }); -} - -const sha256 = (value: string | Buffer): string => createHash("sha256").update(value).digest("hex"); - export function workflowControlCredentialValue( value: string, name: string, @@ -391,26 +261,24 @@ export function workflowControlCredentialValue( return value; } +/** + * The sealed execution snapshot this process runs from, under every name the + * process anchor advertises for it. A native continuation runs the target's + * own workflow and advertises none: treating that project as controller-only + * state would blank every adapter-owned path under it, such as OpenCode's + * run-scoped XDG roots and Kimi's API-key home. + */ function workflowExecutionSnapshotRoots(source: Record): string[] { const roots = new Set(); - const persistedWorkflow = source.ULTRAFUZZ_WORKFLOW_PERSISTED_PATH; - if (persistedWorkflow !== undefined && path.isAbsolute(persistedWorkflow)) { - const workflowsDirectory = path.dirname(persistedWorkflow); - const smithersDirectory = path.dirname(workflowsDirectory); - if (path.basename(workflowsDirectory) === "workflows" && path.basename(smithersDirectory) === ".smithers") { - roots.add(path.dirname(smithersDirectory)); - } - } - for (const name of CONTROLLER_ONLY_ENVIRONMENT_VARIABLES) { - const value = source[name]; - if (value === undefined) continue; - const candidate = environmentPath(value); - for (const marker of ["/dependencies/", "/modules/", "/controls/", "/.smithers/workflows/"]) { - const index = candidate.indexOf(marker); - if (index > 0) roots.add(candidate.slice(0, index)); - } + for (const name of [ + "ULTRAFUZZ_SNAPSHOT_PERSISTED_ROOT", + "ULTRAFUZZ_SNAPSHOT_PROCESS_ROOT", + "ULTRAFUZZ_SNAPSHOT_SOURCE_ROOT" + ]) { + const root = source[name]?.trim(); + if (root && path.isAbsolute(root) && root !== path.parse(root).root) roots.add(root); } - return [...roots].filter((root) => root !== path.parse(root).root); + return [...roots]; } function environmentPath(value: string): string { diff --git a/packages/runtime/src/templates/smithers/agents/kimi.tsx b/packages/runtime/src/templates/smithers/agents/kimi.tsx index ddca25b66..08c59c366 100644 --- a/packages/runtime/src/templates/smithers/agents/kimi.tsx +++ b/packages/runtime/src/templates/smithers/agents/kimi.tsx @@ -161,11 +161,14 @@ export class KimiCode029Agent extends SmithersKimiAgent { return base.onStderrLine?.(line) ?? []; }, onExit: (result) => { - this.issuedSessionId ??= sessionIdFromIndex(this.activeRuntimeHome); + // Session recovery and usage are read back from files the CLI wrote. + // An unreadable or unexpected record leaves them unknown; it must never + // replace the invocation's own result with a telemetry failure. + this.issuedSessionId ??= failOpen(() => sessionIdFromIndex(this.activeRuntimeHome)); // Kimi Code 0.29.1 prints no usage on stdout, so the pinned Smithers // BaseCliAgent falls back to the completed event. Attach this // invocation's own wire-record delta there. - const delta = this.invocationUsage(); + const delta = failOpen(() => this.invocationUsage()); this.pendingFailureUsage = delta === undefined ? undefined : kimiSmithersUsage(delta); const events = base.onExit?.(result) ?? []; if (delta === undefined) return events; @@ -234,11 +237,14 @@ export class KimiCode029Agent extends SmithersKimiAgent { ); } if (executionConfigDir !== undefined && sessionStoreDir !== undefined) { - runtimeHome = createKimiRuntimeHome(executionConfigDir, sessionStoreDir); - seedKimiSessionState(sessionStoreDir, runtimeHome, knownSession); + const seededHome = createKimiRuntimeHome(executionConfigDir, sessionStoreDir); + runtimeHome = seededHome; + seedKimiSessionState(sessionStoreDir, seededHome, knownSession); // Baseline AFTER seeding so a resumed session reports only the tokens - // this invocation adds, never the history it inherited. - usageBaseline = kimiUsageBaseline(runtimeHome); + // this invocation adds, never the history it inherited. Unreadable + // inherited history (for example a wire torn by a killed attempt) + // leaves this invocation's usage unknown instead of failing it. + usageBaseline = failOpen(() => kimiUsageBaseline(seededHome)); } } catch (error) { await command.cleanup?.(); @@ -1508,27 +1514,22 @@ function combineCleanup( }; } +// Runs inside the child's stdout/stderr listeners, where a throw escapes the +// invocation instead of failing it and leaves the task waiting for its timeout. +// Any line that is not an unambiguous resume hint therefore carries no session. function sessionIdFromJsonLine(line: string): string | undefined { const first = firstNonJsonWhitespace(line); if (first === undefined || first !== "{") return undefined; - let value: unknown; - try { - value = parseStrictJson(line, { + const value = failOpen(() => + parseStrictJson(line, { maxBytes: KIMI_WIRE_MAX_LINE_BYTES, maxDepth: KIMI_JSON_MAX_DEPTH, maxItems: KIMI_JSON_MAX_ITEMS, maxProperties: KIMI_JSON_MAX_PROPERTIES - }); - } catch (error) { - throw new Error(`Kimi output JSON is invalid: ${error instanceof Error ? error.message : String(error)}`, { - cause: error - }); - } + }) + ); if (!isRecord(value) || value.type !== KIMI_RESUME_HINT_TYPE) return undefined; - if (typeof value.session_id !== "string") throw new Error("Kimi session.resume_hint session_id must be a string"); - const sessionId = validSessionId(value.session_id); - if (sessionId === undefined) throw new Error("Kimi session.resume_hint session_id is invalid"); - return sessionId; + return typeof value.session_id === "string" ? validSessionId(value.session_id) : undefined; } function firstNonJsonWhitespace(value: string): string | undefined { @@ -1559,6 +1560,14 @@ function isRecord(value: unknown): value is Record { return typeof value === "object" && value !== null && !Array.isArray(value); } +function failOpen(read: () => T): T | undefined { + try { + return read(); + } catch { + return undefined; + } +} + function pathEntryExists(filePath: string): boolean { try { lstatSync(filePath); diff --git a/packages/runtime/src/templates/smithers/agents/pi.tsx b/packages/runtime/src/templates/smithers/agents/pi.tsx index 48ff75343..83287b1b1 100644 --- a/packages/runtime/src/templates/smithers/agents/pi.tsx +++ b/packages/runtime/src/templates/smithers/agents/pi.tsx @@ -99,7 +99,13 @@ export class CompatiblePiAgent extends SmithersPiAgent { return { ...interpreter, onStdoutLine: (line) => { - const observation = observePiLine(usage, line); + // This runs inside the child's stdout listener, where a throw escapes + // the invocation instead of failing it and leaves the task waiting for + // its timeout. A response whose usage Pi reports inconsistently is left + // out whole, so the invocation's usage is then a lower bound. It cannot + // become unknown instead: earlier responses were already reported as + // cumulative usage snapshots, which the engine persists. + const observation = failOpen(() => observePiLine(usage, line)); sessionId = observation?.sessionId ?? sessionId; // A fresh Smithers interpreter sees only the latest authoritative // assistant message and subsequent deltas. Reusing it as an oracle @@ -202,11 +208,9 @@ function observePiLine(totals: PiInvocationUsage, line: string): PiLineObservati if (message?.role !== "assistant") return; const usage = objectRecord(message.usage); if (usage === undefined) return; - const responseId = message.responseId; - if (typeof responseId === "string" && responseId.length > 0) { - if (totals.responseIds.has(responseId)) return; - totals.responseIds.add(responseId); - } + const responseId = + typeof message.responseId === "string" && message.responseId.length > 0 ? message.responseId : undefined; + if (responseId !== undefined && totals.responseIds.has(responseId)) return; const freshInputTokens = piUsageCount(usage.input, "input"); const outputTokens = piUsageCount(usage.output, "output"); const cacheReadTokens = piUsageCount(usage.cacheRead, "cacheRead"); @@ -232,18 +236,24 @@ function observePiLine(totals: PiInvocationUsage, line: string): PiLineObservati if (Math.abs(reportedCostUsd - componentCost) > Math.max(1e-12, Math.abs(reportedCostUsd) * 1e-9)) { throw new Error("Pi assistant adapter-recorded cost total does not equal its cost component sum"); } - totals.messageCount += 1; - totals.freshInputTokens = safePiUsageSum(totals.freshInputTokens, freshInputTokens); - totals.outputTokens = safePiUsageSum(totals.outputTokens, outputTokens); - totals.cacheReadTokens = safePiUsageSum(totals.cacheReadTokens, cacheReadTokens); - totals.cacheWriteTokens = safePiUsageSum(totals.cacheWriteTokens, cacheWriteTokens); - totals.reasoningTokens = safePiUsageSum(totals.reasoningTokens, reasoningTokens); - totals.reportedCostUsd += reportedCostUsd; - if (!Number.isFinite(totals.reportedCostUsd) || totals.reportedCostUsd < 0) { + const next: PiInvocationUsage = { + responseIds: totals.responseIds, + messageCount: totals.messageCount + 1, + freshInputTokens: safePiUsageSum(totals.freshInputTokens, freshInputTokens), + outputTokens: safePiUsageSum(totals.outputTokens, outputTokens), + cacheReadTokens: safePiUsageSum(totals.cacheReadTokens, cacheReadTokens), + cacheWriteTokens: safePiUsageSum(totals.cacheWriteTokens, cacheWriteTokens), + reasoningTokens: safePiUsageSum(totals.reasoningTokens, reasoningTokens), + reportedCostUsd: totals.reportedCostUsd + reportedCostUsd + }; + if (!Number.isFinite(next.reportedCostUsd) || next.reportedCostUsd < 0) { throw new Error("Pi assistant reported cost aggregate is invalid"); } - const cumulativeUsage = reportedPiUsage(totals); + const cumulativeUsage = reportedPiUsage(next); if (cumulativeUsage === undefined) throw new Error("Pi assistant usage aggregate was not recorded"); + // Commit only after every check passed, so a response is never half counted. + Object.assign(totals, next); + if (responseId !== undefined) totals.responseIds.add(responseId); return { progress: { usage: cumulativeUsage, @@ -274,10 +284,10 @@ function safePiUsageSum(left: number, right: number): number { function reportedPiUsage(totals: PiInvocationUsage): PiReportedUsage | undefined { if (totals.messageCount === 0) return undefined; - const inputTokens = safePiUsageSum( - safePiUsageSum(totals.freshInputTokens, totals.cacheReadTokens), - totals.cacheWriteTokens - ); + const inputTokens = totals.freshInputTokens + totals.cacheReadTokens + totals.cacheWriteTokens; + const totalTokens = inputTokens + totals.outputTokens; + // An aggregate outside the safe-integer range is unknown usage, not a failure. + if (!Number.isSafeInteger(totalTokens)) return undefined; return { inputTokens, freshInputTokens: totals.freshInputTokens, @@ -285,7 +295,7 @@ function reportedPiUsage(totals: PiInvocationUsage): PiReportedUsage | undefined cacheReadTokens: totals.cacheReadTokens, cacheWriteTokens: totals.cacheWriteTokens, reasoningTokens: totals.reasoningTokens, - totalTokens: safePiUsageSum(inputTokens, totals.outputTokens), + totalTokens, reportedCostUsd: totals.reportedCostUsd }; } @@ -362,6 +372,14 @@ function objectRecord(value: unknown): Record | undefined { return value && typeof value === "object" && !Array.isArray(value) ? (value as Record) : undefined; } +function failOpen(read: () => T): T | undefined { + try { + return read(); + } catch { + return undefined; + } +} + function applyPiTerminalAnswer( events: T, terminalEvents: unknown, @@ -376,7 +394,11 @@ function applyPiTerminalAnswer( for (const value of Array.isArray(events) ? events : [events]) { const event = objectRecord(value); if (event?.type !== "completed") continue; - if (usage !== undefined) event.usage = usage; + // Smithers' interpreter keeps Pi's raw last-response usage object, from + // which the engine can read only a total. With nothing counted here, that + // lone total would be reported as the invocation's usage. + if (usage === undefined) delete event.usage; + else event.usage = usage; if (terminalError !== undefined) { event.ok = false; event.error = terminalError; diff --git a/packages/runtime/src/templates/smithers/workflows/workflow.tsx b/packages/runtime/src/templates/smithers/workflows/workflow.tsx index ceab70669..d7ca97b86 100644 --- a/packages/runtime/src/templates/smithers/workflows/workflow.tsx +++ b/packages/runtime/src/templates/smithers/workflows/workflow.tsx @@ -38,7 +38,6 @@ const runtimeModule = process.env.ULTRAFUZZ_RUNTIME_MODULE ?? new URL("../../modules/@ultrafuzz/runtime/dist/index.js", import.meta.url).href; const { - artifactContractDefinition, artifactContractSchemaBinding, artifactSchemaRegistry, artifactValidatorSmokeFixturePath, @@ -478,9 +477,10 @@ function compiledTaskSourceIdentity(task: (typeof compiledBaseTasks)[number]) { } function taskSpecsFromCompiled(tasks: typeof compiledBaseTasks) { + const compiledById = new Map(serializedTaskSpecs.map((candidate) => [candidate.id, candidate])); return tasks.map((task) => { const controlPaths = taskWorkflowControlPaths(task.execution.mode, admittedWorkflowControls); - const compiled = serializedTaskSpecs.find((candidate) => candidate.id === task.smithersNodeId); + const compiled = compiledById.get(task.smithersNodeId); const runtimePromptPath = task.renderedPromptPath === undefined ? undefined @@ -1678,13 +1678,13 @@ function renderAgentPrompt(values: { runtimeContext: string; operatorPrompt: str ["operator_prompt", values.operatorPrompt], ["task_prompt", values.taskPrompt] ]); - const rendered = agentPromptTemplate.replace(/\{\{\s*([A-Za-z0-9_]+)\s*\}\}/gu, (match: string, key: string) => - replacements.has(key) ? replacements.get(key)! : match - ); - if (/\{\{\s*[A-Za-z0-9_]+\s*\}\}/u.test(rendered)) { - throw new Error("agent prompt template contains an unresolved variable"); - } - return rendered; + // Check only the trusted template's own placeholders. The inserted prompts may legitimately + // contain literal `{{word}}` text, and rejecting it here would fail every render of the run. + return agentPromptTemplate.replace(/\{\{\s*([A-Za-z0-9_]+)\s*\}\}/gu, (_match: string, key: string) => { + const value = replacements.get(key); + if (value === undefined) throw new Error(`agent prompt template contains an unresolved variable: ${key}`); + return value; + }); } function sourceUsesPinnedBranch(): boolean { @@ -3346,7 +3346,10 @@ function artifactAwareAgent( "AGENT_CONFIG_INVALID", "AGENT_SESSION_LOST", "AGENT_CHECKPOINT_INVALID", - "TASK_ABORTED" + "TASK_ABORTED", + // Agent CLI deadlines. Run synchronization labels timeouts by code only. + "PROCESS_TIMEOUT", + "PROCESS_IDLE_TIMEOUT" ]); // #677: a routed-gateway HTTP 402 (provider credit exhausted) reaches this // normalizer as an anonymous CLI failure because the subprocess boundary @@ -3951,10 +3954,11 @@ function preparationStep(attemptId: string, step: string, run: () => T): T { try { return run(); } catch (error) { - throw new Error( + const wrapped = new Error( `prepare:${attemptId} failed at step ${step}: ${error instanceof Error ? error.message : String(error)}`, { cause: error } ); + throw isNonRetryableFailure(error) ? nonRetryableFailure(wrapped) : wrapped; } } @@ -4002,7 +4006,11 @@ function prepareArtifactMirror( materializePromptSchemas(schemaDirectory, { replaceExisting: replacePromptSchemas }) ); preparationStep(task.attemptId, "assert-task-output-schema-bindings", () => assertTaskOutputSchemaBindings(task)); - preparationStep(task.attemptId, "preflight-json-validator", () => preflightJsonValidator(schemaDirectory)); + // `pinnedSubmodules: "verify"` is the post-agent verify pass. It checks outputs in-process and never + // runs the agent-facing CLI, so a CLI cold start there could only fail its zero-retry task. + if (options.pinnedSubmodules !== "verify") { + preparationStep(task.attemptId, "preflight-json-validator", () => preflightJsonValidator(schemaDirectory)); + } preparationStep(task.attemptId, "assert-task-inputs", () => assertTaskInputs(task, workspaceRoot)); preparationStep(task.attemptId, "materialize-workspace-patch-dependencies", () => materializeWorkspacePatchDependencies(task, workspaceRoot, options.replayWorkspacePatches ?? true, evidenceMode) @@ -4067,6 +4075,9 @@ function prepareArtifactMirror( } function assertTaskOutputSchemaBindings(task: (typeof taskSpecs)[number]): void { + // The verifier validates with this process's schemas and records the planned binding in its marker, + // so the schema content must be the planned one. `validatorBuild` is provenance and is not compared: + // a rebuild of the validator modules must not stop an in-flight run (#921). for (const output of task.outputs) { const binding = artifactContractSchemaBinding( output.contract as Parameters[0] @@ -4075,8 +4086,7 @@ function assertTaskOutputSchemaBindings(task: (typeof taskSpecs)[number]): void binding?.schema_file !== output.schemaFile || binding?.schema_id !== output.schemaId || binding?.schema_sha256 !== output.schemaSha256 || - binding?.schema_bundle_sha256 !== output.schemaBundleSha256 || - binding?.validator_build !== output.validatorBuild + binding?.schema_bundle_sha256 !== output.schemaBundleSha256 ) { throw new Error(`artifact-contract failure: planned schema binding changed for ${output.path}`); } @@ -4084,10 +4094,10 @@ function assertTaskOutputSchemaBindings(task: (typeof taskSpecs)[number]): void } /** - * How long the per-node validator preflight may spend inside the Ultrafuzz CLI. + * How long the validator preflight may spend inside the Ultrafuzz CLI. * * `"ultrafuzz"` resolves to the run-owned trusted launcher, which `composeSmithersCommandPath` puts - * first on PATH, so every node preparation pays the CLI's own cold start: ~2.5 s on an idle box, + * first on PATH, so the preflight pays the CLI's own cold start: ~2.5 s on an idle box, * ~35 s once a dozen agents are building against the same cores. Below that the step reports a bare * `spawnSync ultrafuzz ETIMEDOUT`, which names neither the contention nor a schema, and which cost * the smoke lane two of its three targets in #1026. Roughly five times that measured worst case, @@ -4095,8 +4105,14 @@ function assertTaskOutputSchemaBindings(task: (typeof taskSpecs)[number]): void * inside the attempt. */ const JSON_VALIDATOR_PREFLIGHT_TIMEOUT_MS = 180_000; +// The preflight proves that this process can launch the agent-facing validator, which does not vary +// by task: `materializePromptSchemas` has already digest-checked each workspace's schema copy. One +// success per engine process is enough. Re-spawning it in every prepare and attempt reset only +// added CLI cold starts that could fail an attempt. +let jsonValidatorPreflightPassed = false; function preflightJsonValidator(schemaDirectory: string): void { + if (jsonValidatorPreflightPassed) return; const findings = artifactSchemaRegistry().find( (entry: { filename: string }) => entry.filename === "findings.schema.json" ); @@ -4130,6 +4146,7 @@ function preflightJsonValidator(schemaDirectory: string): void { cause: error }); } + jsonValidatorPreflightPassed = true; } function taskPublishesWorkspacePatch(task: (typeof taskSpecs)[number]): boolean { @@ -5578,6 +5595,14 @@ function assertTaskInputs(task: (typeof taskSpecs)[number], workspaceRoot: strin } function assertTaskDependencyInputs(task: (typeof taskSpecs)[number]): void { + try { + admitTaskDependencyInputs(task); + } catch (error) { + throw nonRetryableFailure(error); + } +} + +function admitTaskDependencyInputs(task: (typeof taskSpecs)[number]): void { const existingAdmission = dependencyArtifactAdmissionsByTask.get(task.attemptId); if (existingAdmission !== undefined) { assertDependencyArtifactAdmissionCurrent(task, existingAdmission); @@ -5704,6 +5729,17 @@ function admittedDependencyArtifactDirs(task: (typeof taskSpecs)[number]): reado function assertDependencyArtifactAdmissionCurrent( task: (typeof taskSpecs)[number], expected: DependencyArtifactAdmission = dependencyArtifactAdmission(task) +): DependencyArtifactAdmission { + try { + return checkDependencyArtifactAdmissionCurrent(task, expected); + } catch (error) { + throw nonRetryableFailure(error); + } +} + +function checkDependencyArtifactAdmissionCurrent( + task: (typeof taskSpecs)[number], + expected: DependencyArtifactAdmission ): DependencyArtifactAdmission { if (expected.task !== task || dependencyArtifactAdmissionsByTask.get(task.attemptId) !== expected) { throw new Error(`artifact-contract failure: dependency admission identity changed ${task.attemptId}`); @@ -5749,6 +5785,22 @@ function assertDependencyArtifactAdmissionCurrent( return expected; } +/** + * A dependency admission failure is deterministic for its consumer: the + * producer's verified artifacts are missing, changed, or unsafe, and a retry + * re-reads the same bytes. Smithers fails an attempt whose error carries + * `details.failureRetryable: false` once, instead of spending the consumer's + * retry budget and agent fallback chain on it (#1144). + */ +function nonRetryableFailure(error: unknown): Error { + const failure = error instanceof Error ? error : new Error(String(error)); + return Object.assign(failure, { details: { failureRetryable: false } }); +} + +function isNonRetryableFailure(error: unknown): boolean { + return (error as { details?: { failureRetryable?: unknown } } | undefined)?.details?.failureRetryable === false; +} + function assertVerifiedDependency( task: (typeof taskSpecs)[number], dependency: string, @@ -5855,15 +5907,16 @@ function assertVerifiedDependency( } seenPaths.add(entry.path); const expected = expectedArtifacts.get(entry.path); + // The marker must name the declared output and its schema content. Its contract digest, bundle + // digest and validator build only record the build that planned the output, so none of them is + // compared, with the declaration or with this build (#921). The bytes stay pinned by the + // marker's sha256 and are revalidated below. if ( expected === undefined || expected.contract !== entry.contract || - expected.contractDigest !== entry.contract_digest || expected.schemaFile !== entry.schema_file || expected.schemaId !== entry.schema_id || expected.schemaSha256 !== entry.schema_sha256 || - expected.schemaBundleSha256 !== entry.schema_bundle_sha256 || - expected.validatorBuild !== entry.validator_build || expected.primary !== entry.primary ) { throw new Error(`verification marker artifact is not a declared output ${entry.path}`); @@ -5880,22 +5933,6 @@ function assertVerifiedDependency( `artifact-contract failure: verified dependency artifact is missing ${entry.path}`, MAX_VERIFIED_ARTIFACT_BYTES ); - const definition = artifactContractDefinition(entry.contract as Parameters[0]); - if (definition.digest !== entry.contract_digest) { - throw new Error(`verified dependency contract changed ${entry.path}`); - } - const currentBinding = artifactContractSchemaBinding( - entry.contract as Parameters[0] - ); - if ( - currentBinding?.schema_file !== entry.schema_file || - currentBinding?.schema_id !== entry.schema_id || - currentBinding?.schema_sha256 !== entry.schema_sha256 || - currentBinding?.schema_bundle_sha256 !== entry.schema_bundle_sha256 || - currentBinding?.validator_build !== entry.validator_build - ) { - throw new Error(`verified dependency schema binding changed ${entry.path}`); - } const artifactSha = createHash("sha256").update(artifactSnapshot.bytes).digest("hex"); if (artifactSha !== entry.sha256) { throw new Error(`verified dependency artifact changed ${entry.path}`); @@ -10110,7 +10147,11 @@ export default smithers((ctx) => { timeoutMs={task.timeoutMs} heartbeatTimeoutMs={task.heartbeatTimeoutMs} retries={task.retries} - retryPolicy={task.retryPolicy} + // The planned chain is the whole retry budget. Smithers' + // default stall verdict would end it after three identical + // failures, before later same-agent attempts or fallback + // profiles run (#1084). + retryPolicy={{ ...task.retryPolicy, maxIdenticalFailures: 0 }} metadata={task.metadata} > {fullTaskPrompt} diff --git a/packages/runtime/src/terminal-report-projection.ts b/packages/runtime/src/terminal-report-projection.ts index 5f70cc42e..a32da8800 100644 --- a/packages/runtime/src/terminal-report-projection.ts +++ b/packages/runtime/src/terminal-report-projection.ts @@ -51,8 +51,61 @@ export function projectTerminalReport(input: TerminalReportProjectionInput): Can if (sourceRunId !== undefined && Reflect.get(reportMetadata, "source_run_id") !== sourceRunId) { throw new Error("Verified final report has a different source run identity"); } + report.run_metadata = withWholeRunSummary(reportMetadata as Record, metadata, state.finished_at); report.completion = completion; return projectCanonicalFinalReport(report, context); } throw new ReportUnavailableError("no successful report-agent output is available"); } + +/** + * The report agent copies a run summary that the host captured when the report task started, so its + * elapsed time and accounting miss the report task itself and anything that finished later. Runtime + * presentations restate them from run.json and the recorded finish time. Tokens, spend, and partial + * pricing move together, because the agent's spend may price only part of the run; a spend the + * whole-run record calls unavailable stays unavailable. A value those records do not provide keeps + * the agent's copy; malformed records are ignored, never thrown. + */ +export function withWholeRunSummary( + runMetadata: Record, + metadata: unknown, + finishedAt: unknown +): Record { + const summary = { ...runMetadata }; + const elapsed = elapsedTime(field(metadata, "created_at"), finishedAt); + if (elapsed !== undefined) summary.elapsed_time = elapsed; + const cumulative = field(field(metadata, "accounting"), "cumulative"); + const models = field(cumulative, "models"); + if (isNonEmptyStringList(models)) summary.models_used = [...models]; + const tokens = field(cumulative, "tokens_used"); + const spend = field(cumulative, "estimated_spend"); + if (availableLabel(tokens) && typeof spend === "string" && spend.trim() !== "") { + summary.tokens_used = tokens; + summary.estimated_spend = spend; + summary.partial_pricing = field(cumulative, "partial_pricing") === true; + } + return summary; +} + +function field(value: unknown, key: string): unknown { + return typeof value === "object" && value !== null && !Array.isArray(value) ? Reflect.get(value, key) : undefined; +} + +function isNonEmptyStringList(value: unknown): value is string[] { + return Array.isArray(value) && value.length > 0 && value.every((entry) => typeof entry === "string" && entry !== ""); +} + +function availableLabel(value: unknown): value is string { + return typeof value === "string" && value.trim() !== "" && value.trim().toLowerCase() !== "unavailable"; +} + +/** Same format as the host's report-start projection: `42.0s`, `5m 07s`, or `6h 02m`. */ +function elapsedTime(createdAt: unknown, finishedAt: unknown): string | undefined { + if (typeof createdAt !== "string" || typeof finishedAt !== "string") return undefined; + const totalSeconds = (Date.parse(finishedAt) - Date.parse(createdAt)) / 1_000; + if (!Number.isFinite(totalSeconds) || totalSeconds < 0) return undefined; + if (totalSeconds < 60) return `${totalSeconds.toFixed(1)}s`; + const totalMinutes = Math.floor(totalSeconds / 60); + if (totalMinutes < 60) return `${String(totalMinutes)}m ${String(Math.floor(totalSeconds % 60)).padStart(2, "0")}s`; + return `${String(Math.floor(totalMinutes / 60))}h ${String(totalMinutes % 60).padStart(2, "0")}m`; +} diff --git a/packages/runtime/src/terminal-report.ts b/packages/runtime/src/terminal-report.ts index 46eb8862a..862b59d1b 100644 --- a/packages/runtime/src/terminal-report.ts +++ b/packages/runtime/src/terminal-report.ts @@ -8,8 +8,6 @@ import { assertRunMetadataDocument, assertRunStateDocument, assertSealedPlannedGraph, - artifactContractDefinition, - artifactContractSchemaBinding, executeSemanticGate, layoutForRunRoot, parseSmithersTaskManifestBytes, @@ -236,7 +234,6 @@ function loadTerminalReportInputs(runRoot: string): TerminalReportInputs { if (state.provenance?.workflow.controlGeneration !== sha256Bytes(authority.workflow_control_seal.bytes)) { throw invalidAuthority("terminal report control generation differs from the current seal"); } - assertCurrentContractBindings(authority); const events = replayEvents(layout, Number.MAX_SAFE_INTEGER).records; const stopped = requireStoppedWorkflowEvent(layout, state, events, authority); const completion = deriveTerminalReportCompletion(authority); @@ -262,23 +259,6 @@ function loadTerminalReportInputs(runRoot: string): TerminalReportInputs { }; } -function assertCurrentContractBindings(authority: VerifiedRunOutputAuthoritySnapshot): void { - const graph = assertSealedPlannedGraph(parseStrictJsonBytes(authority.graph.bytes)); - for (const node of graph.nodes) { - for (const output of node.outputs) { - const binding = artifactContractSchemaBinding(output.contract); - if ( - output.contract_digest !== artifactContractDefinition(output.contract).digest || - (binding !== undefined && - ["schema_file", "schema_id", "schema_sha256", "validator_build"].some( - (field) => Reflect.get(output, field) !== Reflect.get(binding, field) - )) - ) - throw invalidAuthority(`terminal report found schema/control drift for ${node.id}`); - } - } -} - function createTerminalReceipt( inputs: TerminalReportInputs, jsonBytes: Buffer, diff --git a/packages/runtime/src/trusted-cli.ts b/packages/runtime/src/trusted-cli.ts index 68053919d..37905c800 100644 --- a/packages/runtime/src/trusted-cli.ts +++ b/packages/runtime/src/trusted-cli.ts @@ -268,7 +268,6 @@ export function runTrustedJsonValidatorPreflight(input: { layout: RunLayout; tru schemaId: paths.schemaId, schemaSha256: paths.schemaSha256, schemaBundleSha256: metadata.schema_bundle_sha256, - validatorBuild: metadata.validator_build, artifactSha256: paths.fixtureSha256 }); assertTrustedCliLauncher({ layout: input.layout, launcherPath: input.trusted.launcherPath }); @@ -310,11 +309,8 @@ export function assertTrustedCliLauncher(input: { layout: RunLayout; launcherPat // closure on every dispatch. const authenticatedClosurePreflights = new Set(); -function preflightTrustedCliClosure( - closure: TrustedCliClosure, - expected: { validator_build: string; schema_bundle_sha256: string } -): void { - const authenticated = [closure.digest, expected.validator_build, expected.schema_bundle_sha256].join("\0"); +function preflightTrustedCliClosure(closure: TrustedCliClosure, expected: { schema_bundle_sha256: string }): void { + const authenticated = [closure.digest, expected.schema_bundle_sha256].join("\0"); if (authenticatedClosurePreflights.has(authenticated)) return; const paths = closurePreflightPaths(closure); const stdout = execFileSync( @@ -344,7 +340,6 @@ function preflightTrustedCliClosure( schemaId: paths.schemaId, schemaSha256: paths.schemaSha256, schemaBundleSha256: expected.schema_bundle_sha256, - validatorBuild: expected.validator_build, artifactSha256: paths.fixtureSha256 }); authenticatedClosurePreflights.add(authenticated); diff --git a/packages/runtime/src/unverified-report-inputs.ts b/packages/runtime/src/unverified-report-inputs.ts index ff9e12e2c..32c1850fa 100644 --- a/packages/runtime/src/unverified-report-inputs.ts +++ b/packages/runtime/src/unverified-report-inputs.ts @@ -13,6 +13,7 @@ import { reportSchema, type ReportVerification } from "@ultrafuzz/artifacts"; +import { loadGoalSearchCoverageSnapshot } from "./final-report-markdown.js"; import { ReportUnavailableError } from "./report-unavailable.js"; type JsonRecord = Record; @@ -31,6 +32,7 @@ export interface UnverifiedReportInputs { observed: ObservedReportCompletion; verification: ReportVerification; agentReport: JsonRecord; + goalSearchCoverage?: unknown; sources_sha256: string; } @@ -53,10 +55,20 @@ export function readUnverifiedReportInputs(root: string): UnverifiedReportInputs observed, verification: { status: "not-checked", reason_codes: [...reader.reasons].sort() }, agentReport, + goalSearchCoverage: readGoalSearchCoverage(root), sources_sha256: sha256Bytes(Buffer.from(JSON.stringify(reader.sources))) }; } +/** An unreadable census renders as unknown goal-search coverage instead of hiding the report. */ +function readGoalSearchCoverage(root: string): unknown { + try { + return loadGoalSearchCoverageSnapshot(root); + } catch { + return undefined; + } +} + class ReportInputReader { readonly reasons = new Set(["verification-unavailable"]); readonly sources: [string, string][] = []; diff --git a/packages/runtime/src/unverified-report.ts b/packages/runtime/src/unverified-report.ts index 466d93414..8489c9b32 100644 --- a/packages/runtime/src/unverified-report.ts +++ b/packages/runtime/src/unverified-report.ts @@ -14,6 +14,7 @@ import { } from "@ultrafuzz/artifacts"; import { projectCanonicalFinalReport } from "./final-report-markdown.js"; +import { withWholeRunSummary } from "./terminal-report-projection.js"; import { loadCurrentFinalReportSnapshot, publishTerminalReport, @@ -111,11 +112,19 @@ function publishUncheckedPresentation(root: string, file: string, bytes: Buffer) function captureUnverifiedReport(inputs: UnverifiedReportInputs): ReportSnapshot { validateSafeId(inputs.runId, "run ID"); - const projection = projectCanonicalFinalReport({ - ...inputs.agentReport, - verification: inputs.verification, - observed_completion: inputs.observed - }); + const projection = projectCanonicalFinalReport( + { + ...inputs.agentReport, + run_metadata: withWholeRunSummary( + inputs.agentReport.run_metadata as Record, + inputs.metadata, + inputs.state?.finished_at + ), + verification: inputs.verification, + observed_completion: inputs.observed + }, + { goalSearchCoverage: inputs.goalSearchCoverage } + ); const jsonBytes = Buffer.from(`${JSON.stringify(projection.report, null, 2)}\n`, "utf8"); const markdownBytes = Buffer.from(projection.markdown, "utf8"); const generation = sha256Bytes(Buffer.concat([jsonBytes, markdownBytes])); diff --git a/packages/runtime/src/validate.ts b/packages/runtime/src/validate.ts index 8e509bb8f..5bbabe9e7 100644 --- a/packages/runtime/src/validate.ts +++ b/packages/runtime/src/validate.ts @@ -11,7 +11,7 @@ import { validateModelProfiles, type ResolvedConfig } from "@ultrafuzz/config"; -import { loadPromptCatalog, projectPromptDir } from "@ultrafuzz/prompts"; +import { loadPromptCatalog, projectPromptDir, projectPromptsDifferingFromBuiltIns } from "@ultrafuzz/prompts"; import { expandTopology, loadTopology, resolveTopologyPath, type ModelProfileSelection } from "@ultrafuzz/topology"; import type { @@ -202,8 +202,21 @@ function validatePrompts(projectRoot: string): { } try { const catalog = loadPromptCatalog({ projectRoot }); + const differing = projectPromptsDifferingFromBuiltIns(catalog); return { - posture: postureFromDiagnostics("prompts", "project prompt catalog loads and variables are strict", []), + posture: postureFromDiagnostics( + "prompts", + differing.length === 0 + ? "project prompt catalog loads and variables are strict" + : `project prompt catalog loads, but ${String(differing.length)} project prompt(s) differ from the built-in prompt at the same path; ultrafuzz validate --json lists them`, + differing.map((relativePath) => ({ + code: "PROMPT_DIFFERS_FROM_BUILT_IN", + message: `.ultrafuzz/prompts/${relativePath} differs from the built-in prompt at the same path; runs use the project copy, and ultrafuzz init without --force keeps it, so to take the built-in version, delete the file and rerun ultrafuzz init`, + severity: "warning" as const, + source: "prompts", + path: `.ultrafuzz/prompts/${relativePath}` + })) + ), summary: { prompt_dir: promptDir, prompt_count: catalog.orderedIds.length diff --git a/packages/runtime/src/verified-output.ts b/packages/runtime/src/verified-output.ts index beabc6301..e893a540a 100644 --- a/packages/runtime/src/verified-output.ts +++ b/packages/runtime/src/verified-output.ts @@ -9,8 +9,6 @@ import { assertNoSymlinkComponents, assertPathInside, assertRegularFileInside, - artifactContractDefinition, - artifactContractSchemaBinding, boundArtifactValidationWarnings, layoutForRunRoot, parseStrictJsonBytes, @@ -39,7 +37,11 @@ import { type SmithersTaskManifestTask } from "@ultrafuzz/artifacts"; -import { verifyRequiredArtifactsForAttempt, type ArtifactGateAttemptAuthority } from "./artifact-gates.js"; +import { + verifyRequiredArtifactSchemaBinding, + verifyRequiredArtifactsForAttempt, + type ArtifactGateAttemptAuthority +} from "./artifact-gates.js"; import { authenticatedDependencyAdmissionAttemptIds } from "./dependency-admission.js"; import { loadGoalSearchCoverageSnapshot, projectCanonicalFinalReport } from "./final-report-markdown.js"; import { verifySealedTaskManifestSnapshot, type VerifiedSealedTaskManifestSnapshot } from "./workflow-integrity.js"; @@ -369,7 +371,7 @@ function loadFinalizedNodeOutputAuthority(input: LoadVerifiedNodeOutputInput): F const artifactDir = safeResolveInside(layout.artifactsDir, candidate.attemptId, "verified artifact directory"); const publicationSnapshots = readAndBindPublications(artifactDir, plannedNode, documents); - const outputSnapshots = validatePlannedOutputSnapshots(artifactDir, plannedNode, publicationSnapshots); + const outputSnapshots = validatePlannedOutputSnapshots(layout, artifactDir, plannedNode, publicationSnapshots); assertPropertyCampaignEvidencePublications(outputSnapshots, publicationSnapshots); const manifestSeal = finalizedManifestSeal(candidate.state); @@ -1170,15 +1172,28 @@ function readAndBindGateContextFiles( } function validatePlannedOutputSnapshots( + layout: RunLayout, artifactDir: string, plannedNode: PlannedGraphNodeDocument, publications: ReadonlyMap ): VerifiedOutputArtifactSnapshot[] { return plannedNode.outputs.map((output) => { - assertCurrentContractBinding(output); const publication = publications.get(output.path); if (publication === undefined) throw invalidAuthority(`planned output is not published: ${output.path}`); - const validation = validateArtifactContractBytes(output.contract, publication.bytes, publication.absolutePath); + // A schema-backed output is validated against the schema content it was planned with, not this + // build's, so a later rebuild or upgrade cannot turn an already verified output invalid (#921). + const schemaIssues = verifyRequiredArtifactSchemaBinding( + layout, + publication.absolutePath, + output, + publication.bytes + ); + const validation = + schemaIssues.length > 0 + ? { ok: false, issues: schemaIssues, value: undefined } + : output.schema_file === undefined + ? validateArtifactContractBytes(output.contract, publication.bytes, publication.absolutePath) + : { ok: true, issues: [], value: parseStrictJsonBytes(publication.bytes) }; if (!validation.ok) { throw invalidOutput( `verified output failed ${output.contract} validation for ${output.path}: ${validation.issues @@ -1343,27 +1358,6 @@ function assertRunAuthorityIdentity(layout: RunLayout, state: RunState): void { } } -function assertCurrentContractBinding(output: PlannedGraphOutput): void { - const definition = artifactContractDefinition(output.contract); - if (definition.digest !== output.contract_digest) { - throw invalidAuthority(`current contract digest changed for ${output.path}`); - } - const binding = artifactContractSchemaBinding(output.contract); - const actualBinding = schemaBinding(output); - if ( - binding === undefined - ? actualBinding !== undefined - : actualBinding === undefined || - binding.schema_file !== actualBinding.schema_file || - binding.schema_id !== actualBinding.schema_id || - binding.schema_sha256 !== actualBinding.schema_sha256 || - binding.validator_build !== actualBinding.validator_build || - !/^[0-9a-f]{64}$/u.test(actualBinding.schema_bundle_sha256) - ) { - throw invalidAuthority(`current JSON Schema binding changed for ${output.path}`); - } -} - function sameOutputContracts( actual: readonly ArtifactManifestOutputContract[], expected: readonly PlannedGraphOutput[] @@ -1388,17 +1382,6 @@ function withoutSha256(entry: ArtifactVerificationEntry): Omit { - if (output.schema_file === undefined) return undefined; - return Object.freeze({ - schema_file: output.schema_file, - schema_id: output.schema_id!, - schema_sha256: output.schema_sha256!, - schema_bundle_sha256: output.schema_bundle_sha256!, - validator_build: output.validator_build! - }); -} - function concreteNodeIdFromManifest(manifest: ArtifactManifest): string | undefined { const metadata: unknown = manifest.provenance.metadata; return isRecord(metadata) && typeof metadata.concrete_node_id === "string" ? metadata.concrete_node_id : undefined; diff --git a/packages/runtime/src/workflow-control.ts b/packages/runtime/src/workflow-control.ts index b6e38439c..80e147858 100644 --- a/packages/runtime/src/workflow-control.ts +++ b/packages/runtime/src/workflow-control.ts @@ -32,6 +32,8 @@ export interface WorkflowControlProjectionInput { export interface WorkflowControlProjection { state: RunState; changed: boolean; + /** The only change is the lease renewal and concurrency clock that every projection advances. */ + observationOnly: boolean; transitioned: boolean; deadlineExceeded: boolean; recoveryDue: boolean; @@ -225,20 +227,39 @@ export function projectWorkflowControlState(input: WorkflowControlProjectionInpu const controlTransition = controlStateChanged(previous, state); state.last_transition_at = controlTransition ? now : previous.last_transition_at; + // A paused run executes nothing and resuming it resets the deadline, so + // cancelling it at the next observation would bound nothing. const deadlineExceeded = state.workflow_deadline_at !== undefined && input.nowMs >= timestampMs(state.workflow_deadline_at, Number.POSITIVE_INFINITY, "workflow deadline") && - !isTerminalRunStatus(state.status); + !isTerminalRunStatus(state.status) && + state.status !== "paused"; + const changed = JSON.stringify(state) !== JSON.stringify(input.state); return { state, - changed: JSON.stringify(state) !== JSON.stringify(input.state), + changed, + observationOnly: + changed && + JSON.stringify(withoutObservationClock(state)) === JSON.stringify(withoutObservationClock(input.state)), transitioned: controlTransition, deadlineExceeded, recoveryDue }; } +function withoutObservationClock(state: RunState): unknown { + const { renewed_at: _renewedAt, expires_at: _expiresAt, ...lease } = state.controller_lease; + const { + observed_at: _observedAt, + queued_duration_ms: _queuedMs, + active_duration_ms: _activeMs, + idle_duration_ms: _idleMs, + ...concurrency + } = state.concurrency; + return { ...state, controller_lease: lease, concurrency }; +} + function assertExactWorkflowStates( nodeStates: ReadonlyMap, runState: WorkflowRunState | undefined diff --git a/packages/runtime/src/workflow-integrity.ts b/packages/runtime/src/workflow-integrity.ts index 4dcae3d23..fd2167cc5 100644 --- a/packages/runtime/src/workflow-integrity.ts +++ b/packages/runtime/src/workflow-integrity.ts @@ -672,8 +672,7 @@ export function verifyWorkflowControlSnapshot( contents, readBoundedRegularFile(layout.root, layout.statePath, "run state"), executionFiles.find((file) => file.snapshotPath === "controls/plan.json")?.contents, - runtimeStateNodeIds, - true + runtimeStateNodeIds ); } catch (error) { reportDivergence( @@ -1304,14 +1303,17 @@ function publishWorkflowExecutionSnapshot( inode: temporaryStat.ino }; assertSnapshotPublicationBoundary(boundary, "workflow execution snapshot creation"); - for (const [relative, contents] of expectedFiles) writeSnapshotFile(accessRoot, relative, contents, boundary); + const executables = new Set(executablePaths); + for (const [relative, contents] of expectedFiles) { + writeSnapshotFile(accessRoot, relative, contents, executables.has(relative) ? 0o500 : 0o400, boundary); + } for (const [relative, target] of expectedLinks) writeSnapshotLink(accessRoot, relative, target, boundary); - sealSnapshotPermissions(accessRoot, new Set(executablePaths), boundary); + sealSnapshotPermissions(accessRoot, boundary); verifyPublishedWorkflowExecutionSnapshot( accessRoot, expectedFiles, expectedLinks, - new Set(executablePaths), + executables, temporaryLexicalPath ); assertSnapshotPublicationBoundary(boundary, "workflow execution snapshot publication"); @@ -1449,6 +1451,7 @@ function writeSnapshotFile( root: string, relativePath: string, contents: Buffer, + mode: number, boundary: SnapshotPublicationBoundary ): void { snapshotPath(root, relativePath, "workflow execution snapshot file"); @@ -1459,7 +1462,7 @@ function writeSnapshotFile( const descriptor = fs.openSync( destination, fs.constants.O_RDWR | fs.constants.O_CREAT | fs.constants.O_EXCL | (fs.constants.O_NOFOLLOW ?? 0), - 0o400 + mode ); try { let offset = 0; @@ -1468,6 +1471,9 @@ function writeSnapshotFile( if (written <= 0) throw new Error(`workflow execution snapshot write made no progress: ${relativePath}`); offset += written; } + // The umask can clear bits of the creation mode. Setting the final mode through the + // creating descriptor lets this file's single flush make its bytes and mode durable. + fs.fchmodSync(descriptor, mode); fs.fsyncSync(descriptor); const observed = Buffer.alloc(contents.byteLength); let readOffset = 0; @@ -1890,11 +1896,7 @@ function workflowSnapshotEnvironment( }); } -function sealSnapshotPermissions( - root: string, - executablePaths: ReadonlySet, - boundary: SnapshotPublicationBoundary -): void { +function sealSnapshotPermissions(root: string, boundary: SnapshotPublicationBoundary): void { if (root !== boundary.accessRoot || boundary.descriptor === undefined) { throw new Error("workflow execution snapshot permission sealing requires its root descriptor"); } @@ -1907,17 +1909,22 @@ function sealSnapshotPermissions( device: boundary.device, inode: boundary.inode }, - "", - executablePaths + "" ); assertSnapshotPublicationBoundary(boundary, "workflow execution snapshot durability flush"); } +/** + * Make every snapshot directory read-only, deepest first. Files already carry + * their final mode from `writeSnapshotFile` and links have none, so neither is + * reopened here. Publication verification then checks each file's mode, link + * count, and bytes, each link's target, that every directory is read-only, and + * that no entry is missing or unexpected. + */ function sealSnapshotDirectoryPermissions( boundary: SnapshotPublicationBoundary, directory: OpenedSnapshotPublicationDirectory, - relativeDirectory: string, - executablePaths: ReadonlySet + relativeDirectory: string ): void { assertSnapshotPublicationBoundary(boundary, "workflow execution snapshot permission seal"); assertOpenedPublicationDirectoryCurrent(directory, "workflow execution snapshot permission seal"); @@ -1938,70 +1945,36 @@ function sealSnapshotDirectoryPermissions( ) { throw new Error(`workflow execution snapshot entry changed during permission sealing: ${relative}`); } - if (accessed.isSymbolicLink()) continue; - if (accessed.isDirectory()) { - const descriptor = openSnapshotDirectory(candidate); - if (descriptor === undefined) { - throw new Error("workflow execution snapshot directory has no descriptor-rooted permission support"); - } - try { - const opened = fs.fstatSync(descriptor); - if (!opened.isDirectory() || opened.dev !== accessed.dev || opened.ino !== accessed.ino) { - throw new Error(`workflow execution snapshot directory changed while sealing: ${relative}`); - } - const descriptorPath = verifiedSnapshotDescriptorPath( - descriptor, - opened.dev, - opened.ino, - "workflow execution snapshot directory" - ); - if (descriptorPath === undefined) { - throw new Error("workflow execution snapshot directory lost descriptor-rooted permission support"); - } - sealSnapshotDirectoryPermissions( - boundary, - { - accessPath: descriptorPath, - lexicalPath: lexicalCandidate, - descriptor, - device: opened.dev, - inode: opened.ino - }, - relative, - executablePaths - ); - } finally { - fs.closeSync(descriptor); - } - continue; - } - if (!accessed.isFile()) { - throw new Error(`workflow execution snapshot contains a non-regular entry while sealing: ${relative}`); + if (!accessed.isDirectory()) continue; + const descriptor = openSnapshotDirectory(candidate); + if (descriptor === undefined) { + throw new Error("workflow execution snapshot directory has no descriptor-rooted permission support"); } - const descriptor = fs.openSync(candidate, fs.constants.O_RDONLY | (fs.constants.O_NOFOLLOW ?? 0)); try { const opened = fs.fstatSync(descriptor); - if (!opened.isFile() || opened.dev !== accessed.dev || opened.ino !== accessed.ino || opened.nlink !== 1) { - throw new Error(`workflow execution snapshot file changed while sealing: ${relative}`); + if (!opened.isDirectory() || opened.dev !== accessed.dev || opened.ino !== accessed.ino) { + throw new Error(`workflow execution snapshot directory changed while sealing: ${relative}`); } - fs.fchmodSync(descriptor, executablePaths.has(relative) ? 0o500 : 0o400); - fs.fsyncSync(descriptor); - const completed = fs.fstatSync(descriptor); - const completedAccess = fs.lstatSync(candidate); - const completedLexical = fs.lstatSync(lexicalCandidate); - for (const observed of [completedAccess, completedLexical]) { - if ( - !observed.isFile() || - observed.isSymbolicLink() || - observed.dev !== completed.dev || - observed.ino !== completed.ino || - observed.mode !== completed.mode || - observed.nlink !== completed.nlink || - observed.size !== completed.size - ) { - throw new Error(`workflow execution snapshot file changed after sealing: ${relative}`); - } + const descriptorPath = verifiedSnapshotDescriptorPath( + descriptor, + opened.dev, + opened.ino, + "workflow execution snapshot directory" + ); + if (descriptorPath === undefined) { + throw new Error("workflow execution snapshot directory lost descriptor-rooted permission support"); } + sealSnapshotDirectoryPermissions( + boundary, + { + accessPath: descriptorPath, + lexicalPath: lexicalCandidate, + descriptor, + device: opened.dev, + inode: opened.ino + }, + relative + ); } finally { fs.closeSync(descriptor); } @@ -2085,12 +2058,9 @@ function deriveWorkflowControlBindings( >, stateContents: Buffer, planContents: Buffer | undefined, - runtimeStateNodeIds?: readonly string[], - allowHistoricalSchemaBundle = false + runtimeStateNodeIds?: readonly string[] ): WorkflowControlBindings { - const graph = (allowHistoricalSchemaBundle ? assertSealedPlannedGraph : assertPlannedGraph)( - parseStrictJsonBytes(contents.graph) - ); + const graph = assertPlannedGraph(parseStrictJsonBytes(contents.graph)); const expandedGraph = assertExpandedGraphSchema(parseStrictJsonBytes(contents.expanded_graph)); const tasksDocument = parseSmithersTaskManifestBytes(contents.tasks); assertSmithersTaskManifestMatchesPlannedGraph(tasksDocument, graph); diff --git a/packages/runtime/src/workflow-sync.ts b/packages/runtime/src/workflow-sync.ts index b8810c84e..30384cdac 100644 --- a/packages/runtime/src/workflow-sync.ts +++ b/packages/runtime/src/workflow-sync.ts @@ -3,10 +3,10 @@ import fs from "node:fs"; import path from "node:path"; import { isDeepStrictEqual } from "node:util"; import { parseResolvedConfigJsonBytes } from "@ultrafuzz/config"; +import lockfile from "proper-lockfile"; import { ARTIFACT_MANIFEST_FILE, - MAX_REFERENCE_ARTIFACT_MANIFEST_AUTHORITY_BYTES, appendUsageEvents, appendNodeAttempts, appendEvent, @@ -19,16 +19,15 @@ import { assertPathInside, assertRegularFileInside, createNodeState, - createNodeAttemptLedgerEntry, createUsageLedgerEntry, getNodeArtifactDir, + isTerminalRunStatus, layoutForRunRoot, manifestDigest, nodeAttemptLedgerIdentity, queryNodeAttempts, readRegularFileSnapshot, readRunMetadataDocument, - reconcileNodeAttemptLedgerEntry, replayEvents, replayNodeAttempts, readRunState, @@ -123,11 +122,9 @@ import { } from "./smithers.js"; import { runsRootForProject } from "./validate.js"; import { projectWorkflowControlState } from "./workflow-control.js"; -import { runtimeSemanticGateDiagnostics } from "./semantic-gates.js"; import { inspectSmithersAttemptAgentSelection, - reconcileSmithersAttemptAgentSelection, - smithersTaskAgentId + type SmithersAttemptAgentSelection } from "./smithers-attempt-authority.js"; import { isRecord } from "@ultrafuzz/artifacts"; @@ -145,6 +142,7 @@ interface WorkflowInspect { steps: WorkflowStep[]; failedWorkflowTaskIds: string[]; exhaustedLoops: CurrentSmithersInspect["exhaustedLoops"]; + runError?: CurrentSmithersInspect["runError"]; } type SmithersRunStatus = CurrentSmithersInspect["runStatus"]; @@ -174,11 +172,11 @@ interface TerminalWorkflowAttempt { outcome: NodeAttemptOutcome; failureCategory?: NodeAttemptFailureCategory; failureMessage?: string; -} - -interface TerminalAttemptSupersessionContext { - crossedRunActivation: boolean; - supersedingStartedSequence: number; + /** + * A later NodeStarted reused this (node, iteration, attempt) numbering, as a + * reset does, so Smithers' attempt row now describes the replacement. + */ + superseded: boolean; } type SmithersNodeAttemptAuthorities = ReadonlyMap; @@ -331,6 +329,8 @@ interface AttemptWorkflowEvidence { evidence: NodeWorkflowEvidence; source: "agent" | "verifier" | "preparation"; taskId: string; + /** The agent task's own attempt number; a verifier numbers its attempts independently. */ + agentAttempt?: number; } interface ObservedTaskEvidence { @@ -350,10 +350,7 @@ interface NodeFinalization { type PendingNodeAppendEvent = Extract< AppendEventInput, - { - eventType: - "node-artifacts-verified" | "node-artifacts-missing" | "findings-validated" | "artifact-manifest-written"; - } + { eventType: "findings-validated" | "artifact-manifest-written" } >; type PendingNodeEvent = PendingNodeAppendEvent extends infer Event ? Event extends PendingNodeAppendEvent @@ -387,9 +384,35 @@ export interface WorkflowSynchronizationControl { allowMissingWorkflowRun?: boolean; /** Status synchronization authenticates published evidence without taking or repairing control state. */ observeOnly?: boolean; + /** Evidence an observer already authenticated for this run, so the pass does not verify it again. */ + evidence?: LinkedWorkflowEvidence; } +/** + * Synchronization failures an observer reports as warnings: a runner query failed or returned + * unusable output, the observation budget ran out, or the pass could not take its lock. None says + * the run's evidence is invalid; local state stays the last coherent snapshot and the next poll + * retries. + */ +export const TRANSIENT_SYNC_DIAGNOSTIC_CODES: ReadonlySet = new Set([ + "WORKFLOW_INSPECT_FAILED", + "WORKFLOW_INSPECT_INVALID", + "WORKFLOW_EVENTS_FAILED", + "WORKFLOW_EVENTS_INVALID", + "WORKFLOW_TOKEN_EVENTS_FAILED", + "WORKFLOW_TOKEN_EVENTS_INVALID", + "WORKFLOW_SYNC_CANCELLED", + "WORKFLOW_SYNC_DEADLINE_EXCEEDED", + "WORKFLOW_SYNC_LOCK_FAILED" +]); + const MAX_OBSERVATION_SYNC_TIMEOUT_MS = 60_000; +/** + * `smithers events` returns at most this many events, oldest first, and says + * nothing when it stops there. The pinned CLI has no sequence cursor to page + * past it, so synchronization can only report that the rest is missing. + */ +const SMITHERS_EVENTS_LIMIT = 100_000; /** * Optionally bound the state refresh performed before read-only observer @@ -513,7 +536,7 @@ export async function refinalizeControllerFailures( } const eventsSnapshot = await runSmithersInspectionCommand({ - args: ["events", input.workflowRunId, "--limit", "100000", "--json"], + args: ["events", input.workflowRunId, "--limit", String(SMITHERS_EVENTS_LIMIT), "--json"], projectRoot: input.projectRoot, env: inspectionEnvironment }); @@ -594,7 +617,8 @@ export async function refinalizeControllerFailures( (attempt) => attempt.nodeId === task.verifierSmithersNodeId && attempt.retry === verifierAttempt && - attempt.outcome === "succeeded" + attempt.outcome === "succeeded" && + !attempt.superseded ); if (matchingAttempts.length !== 1) { throw new Error(`linked workflow does not contain one exact finished verifier attempt for ${task.attemptId}`); @@ -1038,250 +1062,90 @@ function parseFinishedVerifierOutput( }; } +/** Controls of passes that hold their run's synchronization lock. */ +const exclusiveSynchronizations = new WeakSet(); + /** - * Preserve failed local executor occurrences while their exact selected-agent - * detail still exists. Only a stopped-run reset calls this checkpoint; normal - * continuation, active force resets and success publication keep their existing - * paths. It never finalizes artifacts, projects run state or resolves pricing. + * Two passes over one newly finished node would both finalize it and race on its manifest, state + * and journals. A pass therefore runs only under its run's synchronization lock, and a pass that + * finds the lock held leaves the run to the pass in progress instead of waiting for it. */ -export async function preserveFailedWorkflowAttemptsBeforeReset( - input: SyncRunInput & { inspection: SmithersCommandSnapshot } -): Promise { - const projectRoot = path.resolve(input.projectRoot); - const evidence = await readLinkedWorkflowEvidence(projectRoot, input.runId, { observeOnly: true }); - if (!evidence.ok) throw new Error(evidence.diagnostics.map((entry) => entry.message).join("; ")); - const loaded = loadSynchronizationInputs({ - graph: evidence.verifiedControl.contents.graph, - tasks: evidence.verifiedControl.contents.tasks - }); - if (!loaded.ok) throw new Error(loaded.diagnostics.map((entry) => entry.message).join("; ")); - const environment = linkedWorkflowExecutionEnvironment(evidence, input.env); - const inspect = parseCurrentSmithersInspect(input.inspection, evidence.smithersRunId); - const readEvents = async (): Promise => { - const snapshot = await runSmithersInspectionCommand({ - args: ["events", evidence.smithersRunId, "--limit", "100000", "--json"], - projectRoot, - env: environment - }); - if (!snapshot.ok) throw new Error(workflowSnapshotDiagnostic(snapshot, "WORKFLOW_EVENTS_FAILED").message); - return snapshot; - }; - const eventsSnapshot = await readEvents(); - const events = parseWorkflowEvents(eventsSnapshot.stdout, evidence.smithersRunId); - const allExisting = replayNodeAttempts(evidence.layout).entries; - const recordedSequences = recordedTerminalAttemptSequencesFromEntries(allExisting, evidence.smithersRunId); - const tasksByNodeId = new Map( - loaded.tasks.filter((task) => task.execution.mode === "local").map((task) => [task.smithersNodeId, task]) - ); - const failures = resetTerminalFailureAttempts({ - layout: evidence.layout, - workflowRunId: evidence.smithersRunId, - tasksByNodeId, - events, - recordedSequences - }); - if (failures.length === 0) return; - const terminalSequences = new Set( - failures - .filter((attempt) => !recordedSequences.has(attempt.finishedSequence)) - .map((attempt) => attempt.finishedSequence) - ); - const authorities = await inspectTerminalAttemptAuthorities({ - projectRoot, - workflowRunId: evidence.smithersRunId, - tasks: loaded.tasks, - events, - layout: evidence.layout, - env: environment, - control: {}, - terminalSequences - }); - const forbiddenSecretValues = sensitiveEnvironmentValues( - input.env ?? process.env, - loaded.tasks.flatMap((task) => [ - ...task.execution.agentCredentialEnv, - ...(task.execution.modal?.credentialEnv ?? []) - ]) - ); - const pending = failedResetAttemptInputs({ - layout: evidence.layout, - workflowRunId: evidence.smithersRunId, - controlGeneration: evidence.controlGeneration, - tasksByNodeId, - failures, - authorities, - allExisting, - forbiddenSecretValues - }); - validateFailedResetAttemptInputs({ layout: evidence.layout, pending, allExisting, events }); - await assertStoppedResetAuthorityUnchanged({ - evidence, - projectRoot, - environment, - inspect, - events, - readEvents - }); - appendNodeAttempts(evidence.layout, pending); -} - -function resetTerminalFailureAttempts(input: { - layout: RunLayout; - workflowRunId: string; - tasksByNodeId: ReadonlyMap; - events: WorkflowEvent[]; - recordedSequences: ReadonlySet; -}): TerminalWorkflowAttempt[] { - const relevantEvents = input.events.filter( - (event) => event.type === "RunStarted" || input.tasksByNodeId.has(stringField(event.payload, "nodeId") ?? "") - ); - return terminalWorkflowAttempts(relevantEvents, { - recordedTerminalSequences: input.recordedSequences, - authorizeUnrecordedSuperseded: (attempt, context) => { - const task = input.tasksByNodeId.get(attempt.nodeId); - return ( - context.crossedRunActivation && - task !== undefined && - supersededSuccessfulAttemptHasTraceAuthority( - input.layout, - input.workflowRunId, - task, - attempt, - input.events, - context - ) - ); - } - }).filter((attempt) => attempt.outcome === "failed" || attempt.outcome === "timed-out"); -} - -function failedResetAttemptInputs(input: { - layout: RunLayout; - workflowRunId: string; - controlGeneration: string; - tasksByNodeId: ReadonlyMap; - failures: readonly TerminalWorkflowAttempt[]; - authorities: SmithersNodeAttemptAuthorities; - allExisting: readonly NodeAttemptLedgerEntry[]; - forbiddenSecretValues: readonly string[]; -}): AppendNodeAttemptInput[] { - const existing = new Map(input.allExisting.map((entry) => [nodeAttemptLedgerIdentity(entry), entry])); - return input.failures.flatMap((attempt): AppendNodeAttemptInput[] => { - const task = input.tasksByNodeId.get(attempt.nodeId); - if (task === undefined) throw new Error("failed reset attempt has no sealed task authority"); - const prior = existing.get(JSON.stringify([input.workflowRunId, attempt.finishedSequence])); - if ( - prior === undefined && - inspectSmithersAttemptAgentSelection( - task, - input.authorities.get(smithersNodeAttemptAuthorityKey(attempt.nodeId, attempt.iteration)), - attempt.retry - ) === undefined - ) - return []; - return [ - { - workflowRunId: input.workflowRunId, - // Recompute immutable authority; only an already-recorded agent selection - // can replace mutable detail removed by an interrupted reset. - controlGeneration: input.controlGeneration, - nodeId: task.metadata.node.storageId ?? task.concreteNodeId, - strategyAttemptId: task.attemptId, - iteration: attempt.iteration, - attempt: attempt.retry, - startedEventSequence: attempt.startedSequence, - sourceEventSequence: attempt.finishedSequence, - startedAt: attempt.startedAt, - finishedAt: attempt.finishedAt, - outcome: attempt.outcome, - inputManifestDigest: taskAttemptInputManifestDigest(input.layout, task), - ...(prior === undefined - ? { agent: nodeAttemptAgentProvenance(task, attempt, input.authorities) } - : prior.agent === undefined - ? {} - : { agent: prior.agent }), - ...(attempt.failureCategory === undefined ? {} : { failureCategory: attempt.failureCategory }), - ...(attempt.failureMessage === undefined ? {} : { failureMessage: attempt.failureMessage }), - forbiddenSecretValues: input.forbiddenSecretValues - } - ]; - }); -} - -function validateFailedResetAttemptInputs(input: { - layout: RunLayout; - pending: readonly AppendNodeAttemptInput[]; - allExisting: readonly NodeAttemptLedgerEntry[]; - events: readonly WorkflowEvent[]; -}): void { - const existingByIdentity = new Map(input.allExisting.map((entry) => [nodeAttemptLedgerIdentity(entry), entry])); - const candidates = input.pending.map((entry) => { - const candidate = createNodeAttemptLedgerEntry(input.layout, entry); - const prior = existingByIdentity.get(nodeAttemptLedgerIdentity(candidate)); - if (prior === undefined) return candidate; - const reconciled = reconcileNodeAttemptLedgerEntry(prior, candidate, { failureMessage: entry.failureMessage }); - if (reconciled === undefined) - throw new Error("recorded failed attempt no longer matches its immutable event authority"); - return reconciled; - }); - const proposedEntries = [ - ...input.allExisting, - ...candidates.filter((entry) => !existingByIdentity.has(nodeAttemptLedgerIdentity(entry))) - ]; - for (const candidate of candidates) { - const diagnostics = runtimeSemanticGateDiagnostics({ - schemaFilename: "node-attempt-ledger.schema.json", - document: candidate, - artifactPath: input.layout.attemptLedgerPath, - context: { - attemptLedger: { entries: proposedEntries, sourceEntries: [] }, - eventLog: { - events: input.events.map((event) => ({ - workflow_run_id: event.workflowRunId, - source_event_sequence: event.sourceEventSequence, - timestamp_ms: event.timestampMs, - type: event.type, - payload: event.payload - })) +async function synchronizeExclusively( + input: SyncRunInput, + control: WorkflowSynchronizationControl +): Promise< + { ok: true; value: SyncRunValue; diagnostics: RuntimeDiagnostic[] } | { ok: false; diagnostics: RuntimeDiagnostic[] } +> { + const exclusive = { ...control }; + exclusiveSynchronizations.add(exclusive); + const layoutResult = await checkedRunLayout(path.resolve(input.projectRoot), input.runId); + if (!layoutResult.ok || !fs.existsSync(layoutResult.layout.root)) { + // A missing or invalid run has nothing to lock; the pass reports it. + return synchronizeLinkedWorkflowRun(input, exclusive); + } + const layout = layoutResult.layout; + let release: (() => Promise) | undefined; + try { + release = await tryAcquireSynchronizationLock(layout); + } catch (error) { + // A run root the pass cannot write to (read-only, full, or removed) fails the pass like any other + // write would, so observers report it instead of dying on it. + return { + ok: false, + diagnostics: [ + { + code: "WORKFLOW_SYNC_LOCK_FAILED", + message: `run state synchronization could not take its lock: ${error instanceof Error ? error.message : String(error)}`, + severity: "error", + source: "runtime", + path: layout.root } - } - }); - const errors = diagnostics.filter((diagnostic) => diagnostic.severity === "error"); - if (errors.length > 0) throw new Error(errors.map((entry) => entry.message).join("; ")); + ] + }; + } + if (release === undefined) { + return { + ok: true, + diagnostics: [ + { + code: "WORKFLOW_SYNC_IN_PROGRESS", + message: + "another synchronization of this run is in progress, so this one was skipped; local run state may lag until it finishes", + severity: "info", + source: "workflow", + path: layout.root + } + ], + value: { run_id: layout.runId, run_root: layout.root, status: readRunState(layout).status, synced_nodes: 0 } + }; + } + try { + return await synchronizeLinkedWorkflowRun(input, exclusive); + } finally { + // Failing to remove the lock never fails a finished pass; a lock left behind goes stale. + await release().catch(() => undefined); } } -async function assertStoppedResetAuthorityUnchanged(input: { - evidence: LinkedWorkflowEvidence; - projectRoot: string; - environment: Record; - inspect: CurrentSmithersInspect; - events: readonly WorkflowEvent[]; - readEvents: () => Promise; -}): Promise { - const currentEvents = await input.readEvents(); - const currentInspect = await runSmithersInspectionCommand({ - args: ["inspect", input.evidence.smithersRunId, "--format", "json", "--full-output"], - projectRoot: input.projectRoot, - env: input.environment - }); - if ( - !currentInspect.ok || - !isDeepStrictEqual(parseCurrentSmithersInspect(currentInspect, input.evidence.smithersRunId), input.inspect) || - !isDeepStrictEqual(parseWorkflowEvents(currentEvents.stdout, input.evidence.smithersRunId), input.events) - ) { - throw new Error("stopped workflow changed while preserving failed attempts before reset"); - } - const currentEvidence = await readLinkedWorkflowEvidence(input.projectRoot, input.evidence.layout.runId, { - observeOnly: true - }); - if ( - !currentEvidence.ok || - currentEvidence.smithersRunId !== input.evidence.smithersRunId || - currentEvidence.controlGeneration !== input.evidence.controlGeneration || - !currentEvidence.verifiedControl.contents.tasks.equals(input.evidence.verifiedControl.contents.tasks) - ) { - throw new Error("sealed workflow authority changed while preserving failed attempts before reset"); +/** + * Acquires the lock without waiting. Its target is not the run root: the control lock locks that + * path, and proper-lockfile keys its in-process registry by target path, so a second lock on it + * would release the first. + */ +async function tryAcquireSynchronizationLock(layout: RunLayout): Promise<(() => Promise) | undefined> { + const target = path.join(layout.root, ".workflow-sync"); + try { + return await lockfile.lock(target, { + lockfilePath: `${target}.lock`, + realpath: false, + stale: 300_000, + update: 60_000, + // The default handler throws from a timer and would kill an observer or the eval runner. + onCompromised: () => undefined + }); + } catch (error) { + if (error instanceof Error && "code" in error && error.code === "ELOCKED") return undefined; + throw error; } } @@ -1299,6 +1163,7 @@ export async function synchronizeLinkedWorkflowRun( ): Promise< { ok: true; value: SyncRunValue; diagnostics: RuntimeDiagnostic[] } | { ok: false; diagnostics: RuntimeDiagnostic[] } > { + if (!exclusiveSynchronizations.has(control)) return synchronizeExclusively(input, control); let synchronizationNowMs = synchronizationClock(control); const budgetDiagnostic = synchronizationBudgetDiagnostic(control, synchronizationNowMs); if (budgetDiagnostic !== undefined) { @@ -1325,9 +1190,11 @@ export async function synchronizeLinkedWorkflowRun( }; } - const evidence = await readLinkedWorkflowEvidence(projectRoot, input.runId, { - ...(control.observeOnly === true ? { observeOnly: true } : {}) - }); + const evidence = + control.evidence ?? + (await readLinkedWorkflowEvidence(projectRoot, input.runId, { + ...(control.observeOnly === true ? { observeOnly: true } : {}) + })); if (!evidence.ok) { return { ok: false, diagnostics: evidence.diagnostics }; } @@ -1378,7 +1245,7 @@ export async function synchronizeLinkedWorkflowRun( }; } const eventsSnapshot = await runSmithersInspectionCommand({ - args: ["events", evidence.smithersRunId, "--limit", "100000", "--json"], + args: ["events", evidence.smithersRunId, "--limit", String(SMITHERS_EVENTS_LIMIT), "--json"], projectRoot, env: linkedWorkflowExecutionEnvironment(evidence, input.env), ...inspectionExecutionControl(control, synchronizationNowMs) @@ -1389,7 +1256,7 @@ export async function synchronizeLinkedWorkflowRun( return { ok: false, diagnostics: [postEventsBudgetDiagnostic] }; } const tokenEventsSnapshot = await runSmithersInspectionCommand({ - args: ["events", evidence.smithersRunId, "--type", "token", "--limit", "100000", "--json"], + args: ["events", evidence.smithersRunId, "--type", "token", "--limit", String(SMITHERS_EVENTS_LIMIT), "--json"], projectRoot, env: linkedWorkflowExecutionEnvironment(evidence, input.env), ...inspectionExecutionControl(control, synchronizationNowMs) @@ -1439,9 +1306,23 @@ export async function synchronizeLinkedWorkflowRun( diagnostics: [diagnosticFromError(error, "workflow", "WORKFLOW_TOKEN_EVENTS_INVALID")] }; } - let attemptAuthorities: SmithersNodeAttemptAuthorities; + for (const [stream, consequence, parsed] of [ + ["lifecycle", "node evidence and the attempt ledger can be incomplete", events], + ["token", "usage accounting can be incomplete", tokenEvents] + ] as const) { + const last = parsed.at(-1); + if (parsed.length < SMITHERS_EVENTS_LIMIT || last === undefined) continue; + diagnostics.push({ + code: "WORKFLOW_EVENTS_TRUNCATED", + message: `Smithers returned its ${String(SMITHERS_EVENTS_LIMIT)}-event limit for the ${stream} stream; events after sequence ${String(last.sourceEventSequence)} are not synchronized, so ${consequence}`, + severity: "warning", + source: "workflow" + }); + } + // Attempt-ledger bookkeeping never blocks reconciling node and run state. + let attemptAuthorities: SmithersNodeAttemptAuthorities = new Map(); try { - attemptAuthorities = await inspectTerminalAttemptAuthorities({ + const inspected = await inspectTerminalAttemptAuthorities({ projectRoot, workflowRunId: evidence.smithersRunId, tasks: loaded.tasks, @@ -1450,13 +1331,15 @@ export async function synchronizeLinkedWorkflowRun( env: linkedWorkflowExecutionEnvironment(evidence, input.env), control }); + attemptAuthorities = inspected.authorities; + diagnostics.push(...inspected.diagnostics); } catch (error) { const interrupted = synchronizationInterruptionDiagnostic(error); if (interrupted !== undefined) return { ok: false, diagnostics: [interrupted] }; - return { - ok: false, - diagnostics: [diagnosticFromError(error, "workflow", "WORKFLOW_ATTEMPT_INSPECT_FAILED")] - }; + diagnostics.push({ + ...diagnosticFromError(error, "workflow", "WORKFLOW_ATTEMPT_INSPECT_FAILED"), + severity: "warning" + }); } let syncResult; try { @@ -1540,16 +1423,24 @@ export async function synchronizeLinkedWorkflowRun( if (preAccountingBudgetDiagnostic !== undefined) { return { ok: false, diagnostics: [preAccountingBudgetDiagnostic] }; } - const accountingResult = await synchronizeWorkflowAccounting({ - layout, - workflowRunId: evidence.smithersRunId, - controlGeneration: evidence.controlGeneration, - events: tokenEvents, - tasks: loaded.tasks, - attemptEvents: events, - control, - env: input.env ?? process.env - }); + // Usage accounting is a projection of the usage ledger: a failure is reported + // and retried on the next pass, never allowed to block run status. + let accountingResult: Awaited> = { + changed: false, + available: false + }; + try { + accountingResult = await synchronizeWorkflowAccounting({ + layout, + workflowRunId: evidence.smithersRunId, + controlGeneration: evidence.controlGeneration, + events: tokenEvents, + control, + env: input.env ?? process.env + }); + } catch (error) { + diagnostics.push({ ...diagnosticFromError(error, "workflow", "WORKFLOW_ACCOUNTING_FAILED"), severity: "warning" }); + } if (accountingResult.budgetDiagnostic !== undefined) { return { ok: false, diagnostics: [accountingResult.budgetDiagnostic] }; } @@ -1642,14 +1533,18 @@ export async function synchronizeLinkedWorkflowRun( workflowControl.state.last_transition_at = new Date(observedAtMs).toISOString(); deadlineApplied = true; } catch (error) { - diagnostics.push(smithersDiagnostic(error, "WORKFLOW_DEADLINE_CANCEL_FAILED")); + // The run stays active and the next synchronization requests cancellation again. + diagnostics.push({ ...smithersDiagnostic(error, "WORKFLOW_DEADLINE_CANCEL_FAILED"), severity: "warning" }); } } const preControlMutationBudgetDiagnostic = synchronizationBudgetDiagnostic(control, synchronizationClock(control)); if (preControlMutationBudgetDiagnostic !== undefined) { return { ok: false, diagnostics: [preControlMutationBudgetDiagnostic] }; } - if (workflowControl.changed || deadlineApplied) { + // Only an explicit synchronization of a live run persists a lease renewal on its own. Otherwise every + // status poll, and every observation of a finished run, would rewrite state.json with a new clock. + const persistObservation = control.observeOnly !== true && !isTerminalRunStatus(workflowControl.state.status); + if (deadlineApplied || (workflowControl.changed && (persistObservation || !workflowControl.observationOnly))) { writeRunState(layout, workflowControl.state, { forbiddenSecretValues }); } if (deadlineApplied) { @@ -2026,6 +1921,14 @@ function inspectionExecutionControl( }; } +/** + * Fetch `smithers node` detail for the local occurrences that still need agent + * provenance: those neither recorded nor superseded (a superseded occurrence's + * attempt row now describes its replacement). Synchronization therefore + * inspects only attempts it has not recorded; an unsuperseded pre-agent failure + * is not recorded, so it is inspected again on each pass. A failed inspection is + * a warning that defers the node's occurrences to a later pass. + */ async function inspectTerminalAttemptAuthorities(input: { projectRoot: string; workflowRunId: string; @@ -2034,63 +1937,29 @@ async function inspectTerminalAttemptAuthorities(input: { layout: RunLayout; env: Record; control: WorkflowSynchronizationControl; - terminalSequences?: ReadonlySet; -}): Promise { - const tasksByNodeId = new Map(input.tasks.map((task) => [task.smithersNodeId, task])); +}): Promise<{ authorities: SmithersNodeAttemptAuthorities; diagnostics: RuntimeDiagnostic[] }> { const localTaskNodeIds = new Set( input.tasks.filter((task) => task.execution.mode === "local").map((task) => task.smithersNodeId) ); - const relevantEvents = input.events.filter((event) => { - if (event.type === "RunStarted") return true; - const nodeId = stringField(event.payload, "nodeId"); - return nodeId !== undefined && localTaskNodeIds.has(nodeId); - }); - const recordedTerminalSequences = recordedTerminalAttemptSequences(input.layout, input.workflowRunId); - const pending = terminalWorkflowAttempts(relevantEvents, { - tolerateMissingStarts: true, - recordedTerminalSequences, - authorizeUnrecordedSuperseded: (attempt, context) => { - const task = tasksByNodeId.get(attempt.nodeId); - return ( - context.crossedRunActivation && - task !== undefined && - supersededSuccessfulAttemptHasTraceAuthority( - input.layout, - input.workflowRunId, - task, - attempt, - input.events, - context - ) - ); + const recorded = recordedTerminalAttemptSequences(replayNodeAttempts(input.layout).entries, input.workflowRunId); + const pending = new Map(); + for (const attempt of terminalWorkflowAttempts(input.events)) { + if (localTaskNodeIds.has(attempt.nodeId) && !attempt.superseded && !recorded.has(attempt.finishedSequence)) { + pending.set(smithersNodeAttemptAuthorityKey(attempt.nodeId, attempt.iteration), attempt); } - }).filter( - (attempt) => - tasksByNodeId.get(attempt.nodeId)?.execution.mode === "local" && - (input.terminalSequences === undefined || input.terminalSequences.has(attempt.finishedSequence)) - ); - const grouped = new Map(); - for (const attempt of pending) { - const key = smithersNodeAttemptAuthorityKey(attempt.nodeId, attempt.iteration); - const entries = grouped.get(key) ?? []; - entries.push(attempt); - grouped.set(key, entries); - } - if (grouped.size > input.tasks.length) { - throw new Error("Smithers attempt authority inspection exceeds the sealed task bound"); } const authorities = new Map(); - for (const [key, attempts] of grouped) { + const diagnostics: RuntimeDiagnostic[] = []; + for (const [key, attempt] of pending) { assertSynchronizationBudget(input.control); - const first = attempts[0]!; const snapshot = await runSmithersInspectionCommand({ args: [ "node", - first.nodeId, + attempt.nodeId, "-r", input.workflowRunId, "-i", - String(first.iteration), + String(attempt.iteration), "--format", "json", "--full-output" @@ -2101,15 +1970,15 @@ async function inspectTerminalAttemptAuthorities(input: { }); assertSynchronizationBudget(input.control); if (!snapshot.ok || snapshot.json === undefined) { - throw new Error(workflowSnapshotDiagnostic(snapshot, "WORKFLOW_ATTEMPT_INSPECT_FAILED").message); - } - const task = tasksByNodeId.get(first.nodeId)!; - for (const attempt of attempts) { - inspectSmithersAttemptAgentSelection(task, snapshot.json, attempt.retry); + diagnostics.push({ + ...workflowSnapshotDiagnostic(snapshot, "WORKFLOW_ATTEMPT_INSPECT_FAILED"), + severity: "warning" + }); + continue; } authorities.set(key, snapshot.json); } - return authorities; + return { authorities, diagnostics }; } function smithersNodeAttemptAuthorityKey(nodeId: string, iteration: number): string { @@ -2121,8 +1990,6 @@ async function synchronizeWorkflowAccounting(input: { workflowRunId: string; controlGeneration: string; events: WorkflowEvent[]; - tasks: readonly StoredWorkflowTask[]; - attemptEvents: WorkflowEvent[]; env?: Record; control: WorkflowSynchronizationControl; }): Promise<{ @@ -2137,18 +2004,17 @@ async function synchronizeWorkflowAccounting(input: { if (metadata.workflow.control_generation !== input.controlGeneration) { throw new Error("run.json workflow control generation does not match the linked Smithers run"); } - const storedAccounting = - metadata.accounting === undefined ? undefined : storedAccountingDocument(metadata.accounting, input.workflowRunId); - const existingUsageReplay = replayUsageEvents(input.layout); - if (storedAccounting !== undefined) { - assertAccountingMatchesUsageLedger(storedAccounting, existingUsageReplay.entries, "run.json#$.accounting"); - } + // run.json accounting is a cache derived from usage.jsonl: every pass rebuilds + // it from the ledger and reads the prior copy only for its cached prices. A + // pass stopped between the ledger append and the run.json write therefore + // leaves a stale cache that the next pass replaces (#1138). + const storedAccounting = cachedAccountingDocument(metadata.accounting, input.workflowRunId); const preparedUsage = prepareWorkflowUsageEvents( input.layout, input.workflowRunId, input.controlGeneration, input.events, - existingUsageReplay + replayUsageEvents(input.layout) ); const stateSourceRunId = readRunState(input.layout).source_run_id; @@ -2156,7 +2022,7 @@ async function synchronizeWorkflowAccounting(input: { throw new Error("run.json and state.json disagree on source_run_id"); } if (preparedUsage.entries.length === 0) { - if (storedAccounting !== undefined) { + if (metadata.accounting !== undefined) { throw new Error("run.json accounting cannot exist when the usage ledger is empty"); } return { changed: false, available: false }; @@ -2164,8 +2030,13 @@ async function synchronizeWorkflowAccounting(input: { const storedPricingCatalog = storedAccounting?.pricingCatalog; const storedPricing = storedAccounting?.pricing ?? new Map(); - const previouslyUnresolvedModels = - storedPricingCatalog?.status === "disabled" ? new Set(storedPricingCatalog.unresolved_models) : new Set(); + // A model that a disabled or successfully fetched catalog does not list stays + // unpriced; only an unavailable catalog is fetched again on a later pass. + const previouslyUnresolvedModels = new Set( + storedPricingCatalog === undefined || storedPricingCatalog.status === "unavailable" + ? [] + : storedPricingCatalog.unresolved_models + ); const latestLedgerEvents = workflowEventsFromUsageLedger(latestUsageLedgerEntriesByAttempt(preparedUsage.entries)); const requiredModels = modelsRequiringPricing(latestLedgerEvents); const missingModels = requiredModels.filter( @@ -2242,8 +2113,6 @@ async function synchronizeWorkflowAccounting(input: { accountingChanged || metadata.accounting === undefined ? new Date().toISOString() : metadata.accounting.updated_at }; const nextMetadata = assertRunMetadataDocument({ ...metadata, accounting: nextAccounting }, input.layout.runId); - const validatedAccounting = storedAccountingDocument(nextMetadata.accounting, input.workflowRunId); - assertAccountingMatchesUsageLedger(validatedAccounting, preparedUsage.entries, "proposed run.json#$.accounting"); if (!accountingChanged && preparedUsage.pendingEntries.length === 0) { return { changed: false, available: true }; @@ -2257,12 +2126,7 @@ async function synchronizeWorkflowAccounting(input: { return { changed: false, available: false, budgetDiagnostic: preAccountingMutationBudgetDiagnostic }; } - if (preparedUsage.inputs.length > 0) { - const appended = appendUsageEvents(input.layout, preparedUsage.inputs); - if (!isDeepStrictEqual(appended.replay.entries, preparedUsage.entries)) { - throw new Error("usage ledger changed after its immutable validation snapshot"); - } - } + if (preparedUsage.inputs.length > 0) appendUsageEvents(input.layout, preparedUsage.inputs); if (accountingChanged) { writeRunMetadataDocument(input.layout.runMetadataPath, nextMetadata); } @@ -2384,70 +2248,25 @@ function prepareWorkflowUsageEvents( events: WorkflowEvent[], existingReplay: UsageLedgerReplay ): PreparedUsageLedgerAppend { - const usageEvents = events.filter((event) => event.type === "TokenUsageReported"); - const existingByIdentity = new Map( - existingReplay.entries.map((entry) => [usageLedgerIdentity(entry), entry] as const) - ); - const candidateInputs = new Map(); - const candidates = new Map(); - for (const event of usageEvents) { - const candidateInput = normalizedUsageLedgerInput(workflowRunId, controlGeneration, event); - const candidate = createUsageLedgerEntry(layout, candidateInput); - const identity = usageLedgerIdentity(candidate); - const duplicateCandidate = candidates.get(identity); - if (duplicateCandidate !== undefined) { - if (!isDeepStrictEqual(duplicateCandidate, candidate)) { - throw new Error(`usage event ${identity} appears with conflicting immutable data in the source snapshot`); - } - continue; - } - const existing = existingByIdentity.get(identity); - if (existing !== undefined) { - if (!isDeepStrictEqual(existing, candidate)) { - throw new Error(`usage event ${identity} was already recorded with different immutable data`); - } - continue; - } - candidates.set(identity, candidate); - candidateInputs.set(identity, candidateInput); - } - - const entries = existingReplay.entries.map((entry) => candidates.get(usageLedgerIdentity(entry)) ?? entry); + // A usage event is recorded once by its Smithers identity and never + // re-derived, so a row written by an earlier build cannot become a conflict. + const recorded = new Set(existingReplay.entries.map((entry) => usageLedgerIdentity(entry))); + const entries = [...existingReplay.entries]; const pendingEntries: UsageLedgerEntry[] = []; const inputs: AppendUsageEventInput[] = []; - for (const [identity, candidate] of candidates) { - if (existingByIdentity.has(identity)) continue; - entries.push(candidate); - pendingEntries.push(candidate); - inputs.push(candidateInputs.get(identity)!); - } - const context = { - usageLedger: { entries }, - eventLog: { - events: events.map((event) => ({ - workflow_run_id: event.workflowRunId, - source_event_sequence: event.sourceEventSequence, - timestamp_ms: event.timestampMs, - type: event.type, - payload: event.payload - })) - } - }; - for (const candidate of candidates.values()) { - const diagnostics = runtimeSemanticGateDiagnostics({ - schemaFilename: "usage-ledger.schema.json", - document: candidate, - artifactPath: layout.usageLedgerPath, - context + for (const event of events) { + if (event.type !== "TokenUsageReported") continue; + const identity = usageLedgerIdentity({ + workflow_run_id: event.workflowRunId, + source_event_sequence: event.sourceEventSequence }); - if (diagnostics.some((diagnostic) => diagnostic.severity === "error")) { - throw new Error( - diagnostics - .filter((diagnostic) => diagnostic.severity === "error") - .map((diagnostic) => diagnostic.message) - .join("; ") - ); - } + if (recorded.has(identity)) continue; + recorded.add(identity); + const usageInput = normalizedUsageLedgerInput(workflowRunId, controlGeneration, event); + const entry = createUsageLedgerEntry(layout, usageInput); + entries.push(entry); + pendingEntries.push(entry); + inputs.push(usageInput); } return { inputs, entries, pendingEntries }; } @@ -2461,6 +2280,7 @@ function normalizedUsageLedgerInput( throw new Error("usage ledger input is not an exact TokenUsageReported event for the linked workflow"); } const payload = event.payload; + assertWorkflowEventCorrelation(payload, "TokenUsageReported"); return { workflowRunId, controlGeneration, @@ -2636,6 +2456,16 @@ function emptyAccountingSummary(): AccountingSummary { }; } +function cachedAccountingDocument(value: unknown, workflowRunId: string): StoredAccountingDocument | undefined { + if (value === undefined) return undefined; + try { + return storedAccountingDocument(value, workflowRunId); + } catch { + // An unreadable cache is recomputed from the usage ledger, not trusted. + return undefined; + } +} + function storedAccountingDocument(value: unknown, expectedWorkflowRunId?: string): StoredAccountingDocument { const label = "run.json#$.accounting"; const stored = exactStoredRecord( @@ -4024,6 +3854,7 @@ async function synchronizeTasks(input: { concreteAttempts.push(task.attemptId); taskAttemptsByConcreteNode.set(task.concreteNodeId, concreteAttempts); let retryCount = previous?.retry_count ?? 0; + let currentAttemptExecuted = false; try { const ledger = appendTerminalTaskAttempts({ layout: input.layout, @@ -4033,17 +3864,22 @@ async function synchronizeTasks(input: { events: [...runActivationEvents, ...(eventsByNode.get(task.smithersNodeId) ?? [])].sort( (left, right) => left.sourceEventSequence - right.sourceEventSequence ), - authorityEvents: input.events, - currentAttempt: evidence.attempt, + currentAttempt: attemptEvidence.agentAttempt, currentStatus: patchStatus, finalization, attemptAuthorities: input.attemptAuthorities, forbiddenSecretValues: input.forbiddenSecretValues }); + diagnostics.push(...ledger.diagnostics); retryCount = Math.max(0, ledger.executedAttempts - (ledger.currentAttemptExecuted ? 1 : 0)); + currentAttemptExecuted = ledger.currentAttemptExecuted; changed ||= ledger.appended; } catch (error) { - diagnostics.push(diagnosticFromError(error, "artifacts", "NODE_ATTEMPT_LEDGER_WRITE_FAILED")); + // The next pass retries the append; the node itself still synchronizes. + diagnostics.push({ + ...diagnosticFromError(error, "artifacts", "NODE_ATTEMPT_LEDGER_WRITE_FAILED"), + severity: "warning" + }); } if (previousIsImmutable) { // A successful publication and a terminal invalid-output disposition are @@ -4051,7 +3887,15 @@ async function synchronizeTasks(input: { // evidence, but it cannot re-finalize the node, clear its failure, create // a manifest, or recover it from files that appeared after completion. // Operational failures remain recoverable only when newer Smithers task - // evidence reaches the finalization path above. + // evidence reaches the finalization path above. The retry count is + // attempt bookkeeping, not finalization: once the ledger holds the current + // executed attempt, it follows attempts that the finalizing pass deferred. + if (currentAttemptExecuted && previous?.retry_count !== retryCount) { + updateNodeState(input.layout, task.attemptId, { retry_count: retryCount }, undefined, { + forbiddenSecretValues: input.forbiddenSecretValues + }); + changed = true; + } syncedNodes += 1; continue; } @@ -4092,12 +3936,10 @@ async function synchronizeTasks(input: { } syncedNodes += 1; if (previous?.status !== patchStatus) { - const eventProvenance = eventProvenanceForTask(task); appendEvent(input.layout, { eventType: "node-synced", nodeId: task.attemptId, status: patchStatus, - ...(eventProvenance === undefined ? {} : { provenance: eventProvenance }), payload: { workflow_run_id: input.workflowRunId, workflow_task_id: attemptEvidence.taskId, @@ -4300,19 +4142,7 @@ async function finalizeTerminalTask(input: { } : {}) }, - events: - verifierOutputGate !== undefined - ? [ - { - eventType: "node-artifacts-missing", - status: "failed", - payload: { - output_contracts: input.node.outputs, - missing: verifierOutputGate.missing - } - } - ] - : [] + events: [] }; } @@ -4358,25 +4188,6 @@ async function finalizeTerminalTask(input: { authenticatedGateSnapshots(input.node, verifierAuthority) ); diagnostics.push(...gate.diagnostics); - if (gate.ok) { - events.push({ - eventType: "node-artifacts-verified", - status: "succeeded", - payload: { - output_contracts: input.node.outputs, - missing: gate.missing - } - }); - } else { - events.push({ - eventType: "node-artifacts-missing", - status: "failed", - payload: { - output_contracts: input.node.outputs, - missing: gate.missing - } - }); - } let findingsCount: number | undefined; let artifactManifestSha256: string | undefined; @@ -4542,7 +4353,7 @@ async function finalizeTerminalTask(input: { return { status: "failed", diagnostics, - lastError: errorDiagnostics.map((diagnostic) => diagnostic.message).join("; "), + lastError: errorDiagnostics.map((diagnostic) => durableDiagnosticText(input.layout, diagnostic)).join("; "), provenance: { // A controller publication failure is never an output-contract // success. In particular, the run-state schema forbids publishing @@ -4585,6 +4396,22 @@ async function finalizeTerminalTask(input: { }; } +/** + * Durable failure text names where each error is, relative to the run, so state.json and the attempt + * ledger say which artifact failed without carrying a host path. + */ +function durableDiagnosticText(layout: RunLayout, diagnostic: RuntimeDiagnostic): string { + if (diagnostic.path === undefined) return diagnostic.message; + const pointerStart = diagnostic.path.indexOf("#"); + const filePath = pointerStart === -1 ? diagnostic.path : diagnostic.path.slice(0, pointerStart); + const pointer = pointerStart === -1 ? "" : diagnostic.path.slice(pointerStart); + const relative = path.isAbsolute(filePath) ? path.relative(layout.root, filePath) : filePath; + if (relative === "" || relative === ".." || relative.startsWith(`..${path.sep}`) || path.isAbsolute(relative)) { + return diagnostic.message; + } + return `${relative.split(path.sep).join("/")}${pointer}: ${diagnostic.message}`; +} + function admittedManifestPrerequisiteAttemptIds( task: StoredWorkflowTask, admittedDependencyAttemptIds: readonly string[] @@ -5003,31 +4830,11 @@ function appendNodeEvents( events: PendingNodeEvent[], forbiddenSecretValues: readonly string[] ): void { - const provenance = eventProvenanceForTask(task); for (const event of events) { - appendEvent(layout, { - ...event, - nodeId: task.attemptId, - ...(provenance === undefined ? {} : { provenance }), - forbiddenSecretValues - }); + appendEvent(layout, { ...event, nodeId: task.attemptId, forbiddenSecretValues }); } } -function eventProvenanceForTask(task: StoredWorkflowTask): Record | undefined { - const producerNodeId = task.metadata?.node?.producerNodeId; - const storageId = task.metadata?.node?.storageId; - const dynamic = task.metadata?.node?.dynamic; - if (producerNodeId === undefined && storageId === undefined && dynamic === undefined) return undefined; - return { - producer_node_id: producerNodeId ?? task.concreteNodeId, - concrete_node_id: task.concreteNodeId, - strategy_attempt_id: task.attemptId, - ...(storageId === undefined ? {} : { storage_id: storageId }), - ...(dynamic === undefined ? {} : { dynamic }) - }; -} - function taskAttemptInputManifestDigest(layout: RunLayout, task: StoredWorkflowTask): string { const state = readRunState(layout); return manifestDigest( @@ -5047,58 +4854,64 @@ function appendTerminalTaskAttempts(input: { workflowRunId: string; controlGeneration: string; events: WorkflowEvent[]; - authorityEvents: WorkflowEvent[]; + /** The agent task's own Smithers attempt number, never its verifier's. */ currentAttempt?: number; currentStatus: NodeStatus; finalization: NodeFinalization; attemptAuthorities: SmithersNodeAttemptAuthorities; forbiddenSecretValues: readonly string[]; -}): { appended: boolean; executedAttempts: number; currentAttemptExecuted: boolean } { +}): { appended: boolean; executedAttempts: number; currentAttemptExecuted: boolean; diagnostics: RuntimeDiagnostic[] } { const allExisting = replayNodeAttempts(input.layout).entries; - const observedTerminalAttempts = terminalWorkflowAttempts(input.events, { - recordedTerminalSequences: recordedTerminalAttemptSequencesFromEntries(allExisting, input.workflowRunId), - authorizeUnrecordedSuperseded: (attempt, context) => - context.crossedRunActivation && - supersededSuccessfulAttemptHasTraceAuthority( - input.layout, - input.workflowRunId, - input.task, - attempt, - input.authorityEvents, - context - ) - }); - const terminalAttempts = - input.task.execution.mode === "local" - ? observedTerminalAttempts.filter((attempt) => { - const detail = input.attemptAuthorities.get( - smithersNodeAttemptAuthorityKey(attempt.nodeId, attempt.iteration) - ); - if (detail === undefined) { - throw new Error( - `Smithers attempt authority is unavailable for attempt ${String(attempt.retry)} of ${JSON.stringify(attempt.nodeId)}` - ); - } - return inspectSmithersAttemptAgentSelection(input.task, detail, attempt.retry) !== undefined; - }) - : observedTerminalAttempts; + const recordedByIdentity = new Map(allExisting.map((entry) => [nodeAttemptLedgerIdentity(entry), entry] as const)); + const terminalAttempts = terminalWorkflowAttempts(input.events); const state = readRunState(input.layout); const inputManifestDigest = taskAttemptInputManifestDigest(input.layout, input.task); const manifestPath = path.join(getNodeArtifactDir(input.layout, input.task.attemptId), "artifact-manifest.json"); const outputManifestDigest = fs.existsSync(manifestPath) ? sha256File(manifestPath) : undefined; const existing = allExisting.filter((entry) => entry.strategy_attempt_id === input.task.attemptId); - const existingByIdentity = new Map(allExisting.map((entry) => [nodeAttemptLedgerIdentity(entry), entry] as const)); const sourceEntries = sourceNodeAttempts(input.layout, state.source_run_id, input.task.attemptId); const currentTerminalAttempt = input.currentAttempt === undefined ? undefined - : terminalAttempts.filter((attempt) => attempt.retry === input.currentAttempt).at(-1); + : terminalAttempts.filter((attempt) => attempt.retry === input.currentAttempt && !attempt.superseded).at(-1); const pending: AppendNodeAttemptInput[] = []; - const candidates: NodeAttemptLedgerEntry[] = []; - const candidatesByIdentity = new Map(); + const diagnostics: RuntimeDiagnostic[] = []; let currentAttemptExecuted = false; for (const attempt of terminalAttempts) { const isCurrent = attempt === currentTerminalAttempt; + // An occurrence is recorded once and never re-derived, so later node + // status, finalization or manifest changes cannot turn it into a conflict. + const recorded = recordedByIdentity.get( + nodeAttemptLedgerIdentity({ + workflow_run_id: input.workflowRunId, + source_event_sequence: attempt.finishedSequence + }) + ); + if (recorded !== undefined) { + currentAttemptExecuted ||= isCurrent && recorded.reuse.status === "executed"; + continue; + } + let agent: NodeAttemptAgentProvenance | undefined; + // A superseded occurrence is recorded without agent provenance, because + // Smithers' attempt row now describes its replacement. That includes a + // pre-agent failure, which then counts as an executed attempt. + if (input.task.execution.mode === "local" && !attempt.superseded) { + const detail = input.attemptAuthorities.get(smithersNodeAttemptAuthorityKey(attempt.nodeId, attempt.iteration)); + // Inspection was unavailable this pass; a later pass retries the occurrence. + if (detail === undefined) continue; + try { + const selection = inspectSmithersAttemptAgentSelection(input.task, detail, attempt.retry); + // Smithers never selected an agent-chain rung, so no model ran. + if (selection === undefined) continue; + agent = nodeAttemptAgentProvenance(selection); + } catch (error) { + // Agent provenance is optional evidence: record the occurrence without it. + diagnostics.push({ + ...diagnosticFromError(error, "workflow", "WORKFLOW_ATTEMPT_INSPECT_FAILED"), + severity: "warning" + }); + } + } let outcome = attempt.outcome; let failureCategory = attempt.failureCategory; let failureMessage = attempt.failureMessage; @@ -5119,8 +4932,10 @@ function appendTerminalTaskAttempts(input: { // artifact-validation failure. The attempt ledger is immutable, so wait // for a later synchronization pass with the manifest instead of turning a // successful executor outcome into a permanent phantom failure (#352). - if (outcome === "succeeded" && outputDigest === undefined) continue; - if (isCurrent && ["failed", "timed-out", "canceled"].includes(outcome)) { + // A finished occurrence that a reset superseded before it was recorded has + // no output of its own left to record: the manifest is the replacement's. + if (outcome === "succeeded" && (outputDigest === undefined || attempt.superseded)) continue; + if (isCurrent && (outcome === "failed" || outcome === "timed-out")) { failureMessage = input.finalization.lastError ?? failureMessage; } const reuseSource = @@ -5134,15 +4949,7 @@ function appendTerminalTaskAttempts(input: { if (reuseSource !== undefined && outputDigest === undefined) { outputDigest = reuseSource.outputManifestDigest; } - const reuse = - reuseSource === undefined - ? undefined - : { - status: "reused" as const, - sourceWorkflowRunId: reuseSource.workflowRunId, - sourceEventSequence: reuseSource.sourceEventSequence - }; - const appendInput: AppendNodeAttemptInput = { + pending.push({ workflowRunId: input.workflowRunId, controlGeneration: input.controlGeneration, nodeId: input.task.metadata.node.storageId ?? input.task.concreteNodeId, @@ -5155,96 +4962,35 @@ function appendTerminalTaskAttempts(input: { finishedAt: attempt.finishedAt, outcome, inputManifestDigest, - ...(input.task.execution.mode === "local" - ? { agent: nodeAttemptAgentProvenance(input.task, attempt, input.attemptAuthorities) } - : {}), + ...(agent === undefined ? {} : { agent }), ...(outputDigest === undefined ? {} : { outputManifestDigest: outputDigest }), - ...(reuse === undefined ? {} : { reuse }), + ...(reuseSource === undefined + ? {} + : { + reuse: { + status: "reused" as const, + sourceWorkflowRunId: reuseSource.workflowRunId, + sourceEventSequence: reuseSource.sourceEventSequence + } + }), ...(failureCategory === undefined ? {} : { failureCategory }), ...(failureMessage === undefined ? {} : { failureMessage }), forbiddenSecretValues: input.forbiddenSecretValues - }; - const candidate = createNodeAttemptLedgerEntry(input.layout, appendInput); - const identity = nodeAttemptLedgerIdentity(candidate); - const duplicateCandidate = candidatesByIdentity.get(identity); - if (duplicateCandidate !== undefined) { - if (!isDeepStrictEqual(duplicateCandidate, candidate)) { - throw new Error(`node attempt ${identity} appears with conflicting immutable data in the source snapshot`); - } - currentAttemptExecuted ||= isCurrent && duplicateCandidate.reuse.status === "executed"; - continue; - } - const recordedEntry = existingByIdentity.get(identity); - if (recordedEntry !== undefined) { - const reconciled = reconcileNodeAttemptLedgerEntry(recordedEntry, candidate, { - failureMessage - }); - if (reconciled === undefined) { - throw new Error(`node attempt ${identity} was already recorded with different immutable data`); - } - candidatesByIdentity.set(identity, reconciled); - candidates.push(reconciled); - currentAttemptExecuted ||= isCurrent && recordedEntry.reuse.status === "executed"; - continue; - } - candidatesByIdentity.set(identity, candidate); - candidates.push(candidate); - pending.push(appendInput); - currentAttemptExecuted ||= isCurrent && reuse?.status !== "reused"; - } - const proposedEntries = [ - ...allExisting, - ...candidates.filter((candidate) => !existingByIdentity.has(nodeAttemptLedgerIdentity(candidate))) - ]; - const gateContext = { - attemptLedger: { entries: proposedEntries, sourceEntries }, - eventLog: { - events: input.events.map((event) => ({ - workflow_run_id: event.workflowRunId, - source_event_sequence: event.sourceEventSequence, - timestamp_ms: event.timestampMs, - type: event.type, - payload: event.payload - })) - } - }; - for (const candidate of candidates) { - const diagnostics = runtimeSemanticGateDiagnostics({ - schemaFilename: "node-attempt-ledger.schema.json", - document: candidate, - artifactPath: input.layout.attemptLedgerPath, - context: gateContext }); - if (diagnostics.some((diagnostic) => diagnostic.severity === "error")) { - throw new Error( - diagnostics - .filter((diagnostic) => diagnostic.severity === "error") - .map((diagnostic) => diagnostic.message) - .join("; ") - ); - } + currentAttemptExecuted ||= isCurrent && reuseSource === undefined; } - const results = pending.length === 0 ? [] : appendNodeAttempts(input.layout, pending); - const allEntries = proposedEntries.filter((entry) => entry.strategy_attempt_id === input.task.attemptId); + const appended = appendNodeAttempts(input.layout, pending).filter((result) => result.appended); return { - appended: results.some((result) => result.appended), - executedAttempts: allEntries.filter((entry) => entry.reuse.status === "executed").length, - currentAttemptExecuted + appended: appended.length > 0, + executedAttempts: [...existing, ...appended.map((result) => result.entry)].filter( + (entry) => entry.reuse.status === "executed" + ).length, + currentAttemptExecuted, + diagnostics }; } -function nodeAttemptAgentProvenance( - task: StoredWorkflowTask, - attempt: Pick, - authorities: SmithersNodeAttemptAuthorities -): NodeAttemptAgentProvenance { - const detail = authorities.get(smithersNodeAttemptAuthorityKey(attempt.nodeId, attempt.iteration)); - if (detail === undefined) { - throw new Error( - `Smithers attempt authority is unavailable for attempt ${attempt.retry} of ${JSON.stringify(attempt.nodeId)}` - ); - } - const selection = reconcileSmithersAttemptAgentSelection(task, detail, attempt.retry); +function nodeAttemptAgentProvenance(selection: SmithersAttemptAgentSelection): NodeAttemptAgentProvenance { const agent = selection.profile; return { chain_index: selection.chainIndex, @@ -5259,98 +5005,72 @@ function nodeAttemptAgentProvenance( }; } -function terminalWorkflowAttempts( - events: WorkflowEvent[], - options: { - tolerateMissingStarts?: boolean; - recordedTerminalSequences?: ReadonlySet; - authorizeUnrecordedSuperseded?: ( - attempt: TerminalWorkflowAttempt, - context: TerminalAttemptSupersessionContext - ) => boolean; - } = {} -): TerminalWorkflowAttempt[] { +/** + * Pair each Smithers attempt occurrence's NodeStarted with its terminal event. + * + * An occurrence is identified by its terminal event sequence, so it stays + * recordable after a reset (timetravel / retry-task) restarts the attempt + * numbering: the earlier occurrence is only marked `superseded`, because + * Smithers upserts the attempt row for the replacement (#1099). Unpairable + * events are skipped, never fatal. A start without a terminal was abandoned + * (Smithers cancels in-progress rows at the next RunStarted without an event), + * and a terminal without a live start in the same activation, such as a + * NodeCancelled for an attempt that already ended or never started, is not an + * occurrence (#1139). Nor is a terminal stamped before its start, which no + * ledger row can hold: Smithers stamps a run cancellation's NodeCancelled with + * an instant it takes before its transaction, so a start that commits meanwhile + * can carry a later timestamp. + */ +function terminalWorkflowAttempts(events: readonly WorkflowEvent[]): TerminalWorkflowAttempt[] { const active = new Map< string, Pick >(); - const attempts = new Map(); - const terminalActivations = new Map(); - let activation = 0; + const latestByIdentity = new Map(); + const attempts: TerminalWorkflowAttempt[] = []; for (const event of events) { if (event.type === "RunStarted") { - // Smithers cancels stale in-progress rows before each resumed activation, - // but does not emit NodeCancelled for those abandoned occurrences. Keep - // duplicate starts fail-closed within one activation while allowing the - // next activation to reuse the same durable attempt number. active.clear(); - activation += 1; continue; } - if (event.type !== "NodeStarted" && event.type !== "NodeFinished" && event.type !== "NodeFailed") continue; - const payload = event.payload; - const nodeId = requiredWorkflowEventString(payload.nodeId, `${event.type} nodeId`); - const retry = requiredWorkflowEventCount(payload.attempt, `${event.type} attempt`); - const iteration = requiredWorkflowEventCount(payload.iteration, `${event.type} iteration`); + if (!["NodeStarted", "NodeFinished", "NodeFailed", "NodeCancelled"].includes(event.type)) continue; + const { nodeId, iteration, attempt: retry } = event.payload; + // A NodeCancelled for a node with no live attempt carries `attempt: null`. + if (typeof nodeId !== "string" || !Number.isSafeInteger(iteration) || !Number.isSafeInteger(retry)) continue; const identity = JSON.stringify([nodeId, iteration, retry]); const timestamp = new Date(event.timestampMs).toISOString(); if (event.type === "NodeStarted") { - if (active.has(identity)) throw new Error(`Smithers attempt ${identity} has multiple active NodeStarted events`); - const superseded = attempts.get(identity); - if (superseded !== undefined) { - const terminalActivation = terminalActivations.get(identity); - if (terminalActivation === undefined) { - throw new Error(`Smithers attempt ${identity} is missing its terminal activation authority`); - } - if ( - !options.recordedTerminalSequences?.has(superseded.finishedSequence) && - options.authorizeUnrecordedSuperseded?.(superseded, { - crossedRunActivation: activation > terminalActivation, - supersedingStartedSequence: event.sourceEventSequence - }) !== true - ) { - throw new Error( - `Smithers attempt ${identity} supersedes terminal event ${superseded.finishedSequence} before durable attempt recording` - ); - } - attempts.delete(identity); - terminalActivations.delete(identity); - } + const previous = latestByIdentity.get(identity); + if (previous !== undefined) previous.superseded = true; active.set(identity, { - retry, - iteration, + retry: retry as number, + iteration: iteration as number, nodeId, startedSequence: event.sourceEventSequence, startedAt: timestamp }); continue; } - const terminal = terminalOutcomeForEvent(event); - if (terminal === undefined) continue; const started = active.get(identity); - if (started === undefined) { - if (options.tolerateMissingStarts === true) continue; - throw new Error(`Smithers terminal event has no preceding NodeStarted for ${identity}`); - } + const terminal = terminalOutcomeForEvent(event); + if (started === undefined || terminal === undefined || event.timestampMs < Date.parse(started.startedAt)) continue; active.delete(identity); - attempts.set(identity, { + const occurrence: TerminalWorkflowAttempt = { ...started, finishedSequence: event.sourceEventSequence, finishedAt: timestamp, outcome: terminal.outcome, ...(terminal.failureCategory === undefined ? {} : { failureCategory: terminal.failureCategory }), - ...(terminal.failureMessage === undefined ? {} : { failureMessage: terminal.failureMessage }) - }); - terminalActivations.set(identity, activation); + ...(terminal.failureMessage === undefined ? {} : { failureMessage: terminal.failureMessage }), + superseded: false + }; + latestByIdentity.set(identity, occurrence); + attempts.push(occurrence); } - return [...attempts.values()].sort((left, right) => left.finishedSequence - right.finishedSequence); -} - -function recordedTerminalAttemptSequences(layout: RunLayout, workflowRunId: string): ReadonlySet { - return recordedTerminalAttemptSequencesFromEntries(replayNodeAttempts(layout).entries, workflowRunId); + return attempts; } -function recordedTerminalAttemptSequencesFromEntries( +function recordedTerminalAttemptSequences( entries: readonly NodeAttemptLedgerEntry[], workflowRunId: string ): ReadonlySet { @@ -5359,318 +5079,6 @@ function recordedTerminalAttemptSequencesFromEntries( ); } -function supersededSuccessfulAttemptHasTraceAuthority( - layout: RunLayout, - workflowRunId: string, - task: StoredWorkflowTask, - attempt: TerminalWorkflowAttempt, - events: readonly WorkflowEvent[], - context: TerminalAttemptSupersessionContext -): boolean { - if (attempt.outcome !== "succeeded") return false; - if (!successfulAttemptHasExactTraceAuthority(workflowRunId, task, attempt, events)) return false; - - const current = readRunState(layout).nodes[task.attemptId]; - const manifestPath = path.join(getNodeArtifactDir(layout, task.attemptId), ARTIFACT_MANIFEST_FILE); - let manifestAuthority: - { status: "missing" } | { status: "invalid" } | { status: "valid"; manifest: ArtifactManifest; digest: string }; - try { - const stat = fs.lstatSync(manifestPath); - if (!stat.isFile()) { - manifestAuthority = { status: "invalid" }; - } else { - const bytes = readRegularFileSnapshot(manifestPath, MAX_REFERENCE_ARTIFACT_MANIFEST_AUTHORITY_BYTES); - try { - const value = parseStrictJsonBytes(bytes, { - maxBytes: MAX_REFERENCE_ARTIFACT_MANIFEST_AUTHORITY_BYTES - }); - const validation = validateArtifactManifest(value); - manifestAuthority = validation.ok - ? { status: "valid", manifest: value as ArtifactManifest, digest: sha256Bytes(bytes) } - : { status: "invalid" }; - } catch { - manifestAuthority = { status: "invalid" }; - } - } - } catch (error) { - if ( - typeof error !== "object" || - error === null || - !("code" in error) || - (error as { code?: unknown }).code !== "ENOENT" - ) { - throw error; - } - manifestAuthority = { status: "missing" }; - } - - if (manifestAuthority.status === "missing") { - // A failed task-output disposition is the controller's immutable rejection - // of this otherwise successful executor occurrence. It is exactly the - // state retry-failed may replace at a later run activation. - return !immutableTerminalFinalization(current) || current?.status === "failed"; - } - if (manifestAuthority.status === "invalid") return false; - return currentPublishedReplacementOccurrenceHasAuthority({ - layout, - workflowRunId, - task, - attempt, - events, - context, - current, - manifest: manifestAuthority.manifest, - manifestDigest: manifestAuthority.digest - }); -} - -function successfulAttemptHasExactTraceAuthority( - workflowRunId: string, - task: StoredWorkflowTask, - attempt: TerminalWorkflowAttempt, - events: readonly WorkflowEvent[] -): boolean { - const summaries = events.filter( - (event) => - event.type === "AgentTraceSummary" && - event.sourceEventSequence > attempt.startedSequence && - event.sourceEventSequence < attempt.finishedSequence && - event.workflowRunId === workflowRunId && - event.payload.nodeId === attempt.nodeId && - event.payload.iteration === attempt.iteration && - event.payload.attempt === attempt.retry - ); - if (summaries.length !== 1) return false; - const summaryEvent = summaries[0]!; - const summary = recordField(summaryEvent.payload, "summary"); - if ( - summary?.runId !== workflowRunId || - summary.nodeId !== attempt.nodeId || - summary.iteration !== attempt.iteration || - summary.attempt !== attempt.retry - ) { - return false; - } - const traceStartedAtMs = numberField(summary, "traceStartedAtMs"); - const traceFinishedAtMs = numberField(summary, "traceFinishedAtMs"); - if ( - traceStartedAtMs === undefined || - traceFinishedAtMs === undefined || - traceFinishedAtMs !== summaryEvent.timestampMs || - traceStartedAtMs < Date.parse(attempt.startedAt) || - traceFinishedAtMs > Date.parse(attempt.finishedAt) - ) { - return false; - } - const agentId = stringField(summary, "agentId"); - const model = stringField(summary, "model"); - if (agentId === undefined || model === undefined) return false; - const selections = task.agentChain - .map((profile, chainIndex) => ({ profile, chainIndex })) - .filter(({ chainIndex }) => smithersTaskAgentId(task, chainIndex) === agentId); - if (selections.length !== 1) return false; - const profile = selections[0]!.profile; - return profile.modelName === undefined || profile.modelName === model; -} - -function currentPublishedReplacementOccurrenceHasAuthority(input: { - layout: RunLayout; - workflowRunId: string; - task: StoredWorkflowTask; - attempt: TerminalWorkflowAttempt; - events: readonly WorkflowEvent[]; - context: TerminalAttemptSupersessionContext; - current: NodeState | undefined; - manifest: ArtifactManifest; - manifestDigest: string; -}): boolean { - if (!input.context.crossedRunActivation || input.current?.status !== "succeeded") return false; - const workflow = recordField(input.current.provenance, "workflow"); - const verifierAttempt = numberField(workflow, "attempt"); - if ( - workflow?.run_id !== input.workflowRunId || - workflow.task_id !== input.task.verifierSmithersNodeId || - workflow.agent_task_id !== input.task.smithersNodeId || - workflow.verifier_task_id !== input.task.verifierSmithersNodeId || - workflow.state !== "finished" || - verifierAttempt === undefined - ) { - return false; - } - - const supersedingStarts = input.events.filter( - (event) => - event.sourceEventSequence === input.context.supersedingStartedSequence && - event.type === "NodeStarted" && - event.payload.nodeId === input.attempt.nodeId && - event.payload.iteration === input.attempt.iteration && - event.payload.attempt === input.attempt.retry - ); - if (supersedingStarts.length !== 1) return false; - const supersedingStart = supersedingStarts[0]!; - const verifierIteration = input.task.metadata.loop.index; - const verifierStarts = input.events.filter( - (event) => - event.sourceEventSequence > supersedingStart.sourceEventSequence && - event.type === "NodeStarted" && - event.payload.nodeId === input.task.verifierSmithersNodeId && - event.payload.iteration === verifierIteration && - event.payload.attempt === verifierAttempt && - new Date(event.timestampMs).toISOString() === input.current?.started_at - ); - if (verifierStarts.length !== 1) return false; - const verifierStart = verifierStarts[0]!; - const verifierTerminals = input.events.filter( - (event) => - event.sourceEventSequence > verifierStart.sourceEventSequence && - event.type === "NodeFinished" && - event.payload.nodeId === input.task.verifierSmithersNodeId && - event.payload.iteration === verifierIteration && - event.payload.attempt === verifierAttempt && - new Date(event.timestampMs).toISOString() === input.current?.finished_at - ); - if (verifierTerminals.length !== 1) return false; - const verifierTerminal = verifierTerminals[0]!; - - // The occurrence that first reuses a historical attempt identity may itself - // be abandoned at the next activation. Bind authority to the final producer - // occurrence before the exact published verifier instead of assuming that - // the reused occurrence must be the publisher. Every intervening abandoned - // producer must cross a RunStarted boundary without a terminal event; this - // keeps overlapping starts and unrelated completed occurrences fail-closed. - const producerStarts = input.events - .filter( - (event) => - event.sourceEventSequence >= supersedingStart.sourceEventSequence && - event.sourceEventSequence < verifierStart.sourceEventSequence && - event.type === "NodeStarted" && - event.payload.nodeId === input.attempt.nodeId && - event.payload.iteration === input.attempt.iteration - ) - .sort((left, right) => left.sourceEventSequence - right.sourceEventSequence); - const replacementStart = producerStarts.at(-1); - if (replacementStart === undefined) return false; - let activeProducerAttempt: number | undefined = input.attempt.retry; - for (const event of input.events) { - if ( - event.sourceEventSequence <= supersedingStart.sourceEventSequence || - event.sourceEventSequence >= replacementStart.sourceEventSequence - ) { - continue; - } - if (event.type === "RunStarted") { - activeProducerAttempt = undefined; - continue; - } - if (event.payload.nodeId !== input.attempt.nodeId || event.payload.iteration !== input.attempt.iteration) { - continue; - } - if (event.type === "NodeStarted") { - if (activeProducerAttempt !== undefined) return false; - const retry = numberField(event.payload, "attempt"); - if (retry === undefined || !Number.isSafeInteger(retry) || retry < input.attempt.retry) return false; - activeProducerAttempt = retry; - continue; - } - if (event.type === "NodeFinished" || event.type === "NodeFailed") return false; - } - if (replacementStart !== supersedingStart && activeProducerAttempt !== undefined) return false; - const replacementRetry = numberField(replacementStart.payload, "attempt"); - if ( - replacementRetry === undefined || - !Number.isSafeInteger(replacementRetry) || - replacementRetry < input.attempt.retry - ) { - return false; - } - const nextBoundarySequence = input.events - .filter( - (event) => - event.sourceEventSequence > replacementStart.sourceEventSequence && - (event.type === "RunStarted" || - (event.type === "NodeStarted" && - event.payload.nodeId === input.attempt.nodeId && - event.payload.iteration === input.attempt.iteration)) - ) - .map((event) => event.sourceEventSequence) - .sort((left, right) => left - right)[0]; - const replacementTerminals = input.events.filter( - (event) => - event.sourceEventSequence > replacementStart.sourceEventSequence && - (nextBoundarySequence === undefined || event.sourceEventSequence < nextBoundarySequence) && - (event.type === "NodeFinished" || event.type === "NodeFailed") && - event.payload.nodeId === input.attempt.nodeId && - event.payload.iteration === input.attempt.iteration && - event.payload.attempt === replacementRetry - ); - if (replacementTerminals.length !== 1 || replacementTerminals[0]!.type !== "NodeFinished") return false; - const replacementTerminal = replacementTerminals[0]!; - const replacementAttempt: TerminalWorkflowAttempt = { - retry: replacementRetry, - iteration: input.attempt.iteration, - nodeId: input.attempt.nodeId, - startedSequence: replacementStart.sourceEventSequence, - finishedSequence: replacementTerminal.sourceEventSequence, - startedAt: new Date(replacementStart.timestampMs).toISOString(), - finishedAt: new Date(replacementTerminal.timestampMs).toISOString(), - outcome: "succeeded" - }; - if (!successfulAttemptHasExactTraceAuthority(input.workflowRunId, input.task, replacementAttempt, input.events)) { - return false; - } - - if (replacementTerminal.sourceEventSequence >= verifierStart.sourceEventSequence) return false; - // A finished producer may leave its verifier pending until a later run - // activation. Only a new producer occurrence before that verifier, or a - // boundary/restart inside the verifier occurrence itself, breaks the link. - if ( - input.events.some( - (event) => - event.sourceEventSequence > replacementTerminal.sourceEventSequence && - event.sourceEventSequence < verifierTerminal.sourceEventSequence && - event.type === "NodeStarted" && - event.payload.nodeId === input.attempt.nodeId && - event.payload.iteration === input.attempt.iteration - ) || - input.events.some( - (event) => - event.sourceEventSequence > verifierStart.sourceEventSequence && - event.sourceEventSequence < verifierTerminal.sourceEventSequence && - (event.type === "RunStarted" || - (event.type === "NodeStarted" && - event.payload.nodeId === input.task.verifierSmithersNodeId && - event.payload.iteration === verifierIteration)) - ) - ) { - return false; - } - - const outputContracts = recordField(input.current.provenance, "output_contracts"); - if ( - outputContracts?.ok !== true || - !Array.isArray(outputContracts.missing) || - outputContracts.missing.length !== 0 || - outputContracts.artifact_manifest_sha256 !== input.manifestDigest - ) { - return false; - } - const markerDigest = input.manifest.provenance.verification_marker_sha256; - if (markerDigest === undefined) return false; - const expectedProvenance = JSON.parse( - JSON.stringify({ - run_id: input.layout.runId, - ...artifactProvenance(input.task, input.workflowRunId, markerDigest) - }) - ) as ArtifactProvenance; - return ( - input.manifest.run_id === input.layout.runId && - input.manifest.node_id === input.task.attemptId && - input.manifest.producer_node_id === (input.task.metadata.node.producerNodeId ?? input.task.attemptId) && - Date.parse(input.manifest.created_at) >= Date.parse(input.current.finished_at ?? "") && - isDeepStrictEqual(input.manifest.provenance, expectedProvenance) - ); -} - function terminalOutcomeForEvent(event: WorkflowEvent): | { outcome: NodeAttemptOutcome; @@ -5683,10 +5091,15 @@ function terminalOutcomeForEvent(event: WorkflowEvent): return { outcome: "succeeded" }; case "NodeFailed": { const failureMessage = errorText(event.payload.error); - return errorLooksLikeTimeout(event.payload.error) + return workflowErrorIsTimeout(event.payload.error) ? { outcome: "timed-out", failureCategory: "timeout", ...(failureMessage ? { failureMessage } : {}) } : { outcome: "failed", failureCategory: "executor-error", ...(failureMessage ? { failureMessage } : {}) }; } + case "NodeCancelled": { + // Smithers names the cancellation: run-cancelled, unmounted or aborted. + const reason = stringField(event.payload, "reason"); + return { outcome: "canceled", failureCategory: "canceled", ...(reason ? { failureMessage: reason } : {}) }; + } default: return undefined; } @@ -5714,23 +5127,15 @@ function finalizationFailureCategory( if (outcome === "timed-out") { return "timeout"; } - if (outcome === "canceled") { - return "canceled"; - } if (outcome !== "failed") { return undefined; } - if (finalization.diagnostics.some((diagnostic) => diagnostic.code === "FINDINGS_VALIDATION_FAILED")) { - return "invalid-output"; - } - if ( - finalization.diagnostics.some( - (diagnostic) => diagnostic.code === "ARTIFACT_MANIFEST_WRITE_FAILED" || diagnostic.code.includes("ARTIFACT") - ) - ) { - return "artifact-validation"; - } - return "unknown"; + // Only a finished executor occurrence reaches this overlay, so its node + // failed after the executor finished: a verifier or artifact-gate rejection, + // or a cancellation while the verifier ran. + return finalization.diagnostics.some((diagnostic) => diagnostic.code === "FINDINGS_VALIDATION_FAILED") + ? "invalid-output" + : "artifact-validation"; } function reusedSourceAttempt(input: { @@ -5869,13 +5274,14 @@ function completionEvidenceForTask( if (agentEvidence === undefined) { return undefined; } + const agentAttempt = agentEvidence.attempt === undefined ? {} : { agentAttempt: agentEvidence.attempt }; if (agentEvidence.status !== "succeeded") { - return { evidence: agentEvidence, source: "agent", taskId: task.smithersNodeId }; + return { evidence: agentEvidence, source: "agent", taskId: task.smithersNodeId, ...agentAttempt }; } if (verifierEvidence === undefined) { return undefined; } - return { evidence: verifierEvidence, source: "verifier", taskId: task.verifierSmithersNodeId }; + return { evidence: verifierEvidence, source: "verifier", taskId: task.verifierSmithersNodeId, ...agentAttempt }; } function evidenceFromStep(step: WorkflowStep): NodeWorkflowEvidence { @@ -5929,8 +5335,7 @@ function evidenceFromEvents(events: WorkflowEvent[]): NodeWorkflowEvidence | und break; case "NodeFailed": { const error = errorText(payload.error); - const timedOut = - evidence?.timedOut === true || errorLooksLikeTimeout(payload.error) || errorLooksLikeTimeout(error); + const timedOut = evidence?.timedOut === true || workflowErrorIsTimeout(payload.error); evidence = { ...evidence, status: timedOut ? "timed-out" : "failed", @@ -6092,11 +5497,12 @@ function nonBlockingRuntimeNodeIds(graph: PlannedGraph): ReadonlySet { return ids; } -// A workflow that ends terminally failed while every durable node is still -// non-terminal is unrecoverable by node-level retry: there is nothing to reset -// and the next resume re-finalizes identically. That is a defect in failure -// attribution, so name the workflow tasks the failure was charged to instead of -// leaving the run indistinguishable from an idle one. +// A workflow that ends terminally failed while no durable node failed stopped on +// something no durable node owns: a run-level runner error (for example +// WORKFLOW_RENDER_FAILED) or a failed workflow task outside the durable graph. +// Name the failing workflow tasks and the runner's own error so the run is not +// indistinguishable from an idle one. Recovery is a same-id resume once that +// cause is removed (smithers-terminal-resume.integration.test.ts). function unattributedTerminalWorkflowFailure( inspect: WorkflowInspect, nodeStatuses: Map, @@ -6114,11 +5520,17 @@ function unattributedTerminalWorkflowFailure( return undefined; } const failedWorkflowTasks = inspect.failedWorkflowTaskIds; + const runError = inspect.runError; + // Message only: the durable event payload built from `details` stays ids-only. + const runErrorText = + runError === undefined + ? "" + : `; workflow run error${runError.code === undefined ? "" : ` ${runError.code}`}: ${runError.message}`; return { code: "WORKFLOW_TERMINAL_WITHOUT_FAILED_NODE", message: `workflow run ended ${workflowState} with no failed durable node; failing workflow task(s): ${ failedWorkflowTasks.length === 0 ? "unreported" : failedWorkflowTasks.join(", ") - }`, + }${runErrorText}`, severity: "error", source: "workflow", details: { @@ -6321,21 +5733,21 @@ function parseInspectSnapshot(snapshot: SmithersCommandSnapshot, expectedWorkflo runState: current.runState, steps: current.nodes.map((node) => ({ id: node.nodeId, state: node.state, attempt: node.attempt })), failedWorkflowTaskIds: [...failedWorkflowTaskIds].sort(), - exhaustedLoops: current.exhaustedLoops + exhaustedLoops: current.exhaustedLoops, + ...(current.runError === undefined ? {} : { runError: current.runError }) }; } /** - * The closed-world key contract Ultrafuzz enforces on the pinned runner's - * `TokenUsageReported` payload, which `synchronizeLinkedWorkflowRun` reads on - * every status, state, diagnose and evals row sync. Exported so a test can diff - * it against the engine's own emitters: the check is exact-key and re-throws - * everywhere except `stats`, so a key added upstream takes out synchronization - * for every run on its first agent task -- which is exactly what 0.35.0's - * `freshInputTokens` and `costUsd` did. + * The keys of the pinned runner's `TokenUsageReported` payload. Synchronization + * validates each accounting field it reads and ignores any other key, so a key + * added upstream no longer takes out synchronization the way 0.35.0's + * `freshInputTokens` and `costUsd` did. A test diffs this list against the + * engine's own emitters, so a new usage field is noticed at the pin bump + * rather than silently left out of accounting. */ export const CURRENT_SMITHERS_TOKEN_EVENT_KEY_CONTRACT = { - allowed: [ + known: [ "type", "runId", "nodeId", @@ -6421,16 +5833,14 @@ function parseWorkflowEvents(stdout: string, expectedWorkflowRunId: string): Wor return events; } +/** + * Usage fields are validated where the usage ledger records them, inside the + * accounting pass, so a malformed `TokenUsageReported` event is an accounting + * warning rather than a failure to read the run's events. + */ function validateSmithersEventPayload(type: string, payload: Record, lineNumber: number): void { const label = `Smithers ${type} payload at line ${lineNumber}`; - const attemptEventTypes = new Set([ - "NodeStarted", - "NodeFinished", - "NodeFailed", - "NodeRetrying", - "TokenUsageReported" - ]); - if (attemptEventTypes.has(type)) { + if (["NodeStarted", "NodeFinished", "NodeFailed", "NodeRetrying"].includes(type)) { requiredWorkflowEventString(payload.nodeId, `${label} nodeId`); requiredWorkflowEventCount(payload.iteration, `${label} iteration`); requiredWorkflowEventCount(payload.attempt, `${label} attempt`); @@ -6438,21 +5848,6 @@ function validateSmithersEventPayload(type: string, payload: Record errorLooksLikeTimeout(entry)); +/** + * Smithers reports its deadlines that can fail an Ultrafuzz task with typed + * codes: the engine's task and heartbeat watchdogs, and the process driver's + * total and idle timers for agent CLIs. Message, stack, and cause text are not + * classification input: a validator preflight that failed in 2ms mentions + * "timeout" in its message, and node ids can contain the word (#1144). A + * deadline reported only as text, such as the Modal provider's cloud-node + * deadline, is therefore labelled failed. + */ +const WORKFLOW_TIMEOUT_ERROR_CODES: ReadonlySet = new Set([ + "TASK_TIMEOUT", + "TASK_HEARTBEAT_TIMEOUT", + "PROCESS_TIMEOUT", + "PROCESS_IDLE_TIMEOUT" +]); + +function workflowErrorIsTimeout(error: unknown): boolean { + return isRecord(error) && typeof error.code === "string" && WORKFLOW_TIMEOUT_ERROR_CODES.has(error.code); } /** diff --git a/packages/runtime/test/agent-adapter-boundaries.test.ts b/packages/runtime/test/agent-adapter-boundaries.test.ts index c8899c0f3..2bb3fbd5f 100644 --- a/packages/runtime/test/agent-adapter-boundaries.test.ts +++ b/packages/runtime/test/agent-adapter-boundaries.test.ts @@ -1,6 +1,5 @@ import assert from "node:assert/strict"; import { temporaryRoot } from "./temporary-root.js"; -import crypto from "node:crypto"; import { existsSync, mkdirSync, readFileSync, readdirSync, rmSync, writeFileSync } from "node:fs"; import path from "node:path"; import test from "node:test"; @@ -10,19 +9,12 @@ import * as ts from "typescript"; type OrchestratorResponsibility = "argv-construction" | "filesystem-walking" | "output-interpretation" | "session-handling" | "token-accounting"; -type ResponsibilityPolicy = { - classifiedSourceSha256: string; +type AdapterPolicy = { + purpose: "adapter" | "data-governance" | "provider-home" | "registry" | "strict-input" | "toml"; responsibilities: readonly OrchestratorResponsibility[]; upstreamIssues: readonly string[]; }; -type SourcePolicy = { - maxLines: number; - maxSyntaxNodes: number; - purpose: "adapter" | "data-governance" | "provider-home" | "registry" | "strict-input" | "toml"; - sourceSha256: string; -}; - type SourceUnit = { ast: ts.SourceFile; source: string; @@ -82,38 +74,21 @@ const TOKEN_ACCOUNTING_SIGNALS = new Set([ "total_tokens" ]); -// This deliberately duplicates each exact source fingerprint in a separate, -// classification-owned policy. A source edit must update both the structural -// policy below and this responsibility review, even when the reviewer decides -// that the declared responsibility set remains unchanged. -const responsibilityPolicies: Record = { - "claude.tsx": { - classifiedSourceSha256: "6d8346e882f9ed1e5274d19d6e7cacf27e2743e489a3b931b9ee0488d62ffe76", - responsibilities: [], - upstreamIssues: [] - }, +// Every .ts/.tsx source under the adapter tree declares what it is for and +// which orchestrator responsibilities it owns. Only registered adapters may own +// responsibilities, and each one must link the upstream gap that forces it. +const adapterPolicies: Record = { + "claude.tsx": { purpose: "adapter", responsibilities: [], upstreamIssues: [] }, "codex.tsx": { - classifiedSourceSha256: "a44eb5c47e6374457476a86420eca0c5ff23616fc8201637f0139d37e13b92f5", + purpose: "adapter", responsibilities: ["argv-construction", "session-handling"], upstreamIssues: ["https://github.com/smithersai/smithers/issues/1622"] }, - "deepseek.tsx": { - classifiedSourceSha256: "ea7c6eec70883126e3ee9588b6d3349169652688756b519f1d49c3b5b794e887", - responsibilities: ["output-interpretation", "token-accounting"], - upstreamIssues: ["https://github.com/smithersai/smithers/issues/1624"] - }, - "environment.tsx": { - classifiedSourceSha256: "72d93b360e5b969e00646cfa39a24cac8a3720ce2aef1c7469c51038ef0206cd", - responsibilities: [], - upstreamIssues: [] - }, - "index.tsx": { - classifiedSourceSha256: "ce5f94b3bf12ae40c5b59ebd587a77d1e80e532d92c785f3353d272e627d79e4", - responsibilities: [], - upstreamIssues: [] - }, + "deepseek.tsx": { purpose: "adapter", responsibilities: [], upstreamIssues: [] }, + "environment.tsx": { purpose: "data-governance", responsibilities: [], upstreamIssues: [] }, + "index.tsx": { purpose: "registry", responsibilities: [], upstreamIssues: [] }, "kimi.tsx": { - classifiedSourceSha256: "1551f53080570f026aa8f6f60d533ef51abe8bac453eb0610e0ef11a9f974530", + purpose: "adapter", responsibilities: [ "argv-construction", "filesystem-walking", @@ -126,13 +101,9 @@ const responsibilityPolicies: Record = { "https://github.com/smithersai/smithers/issues/1626" ] }, - "opencode.tsx": { - classifiedSourceSha256: "7d4e22b674e06b00cd537c531c0e7e95d800ffc07b50d02b566a5fd029d0b489", - responsibilities: [], - upstreamIssues: [] - }, + "opencode.tsx": { purpose: "adapter", responsibilities: [], upstreamIssues: [] }, "openrouter.tsx": { - classifiedSourceSha256: "834b8d1893f1a6b7da9667cd5f55ecd7fd5e7dc9179ab2b307562a9913f410b4", + purpose: "adapter", responsibilities: ["argv-construction", "output-interpretation", "session-handling", "token-accounting"], upstreamIssues: [ "https://github.com/monad-developers/ultrafuzz/issues/1006", @@ -142,7 +113,7 @@ const responsibilityPolicies: Record = { ] }, "pi.tsx": { - classifiedSourceSha256: "1b9e81f7d79a7c7f551c1b6d7d1f794e2d49ee698b328eee4a872b914bfc8acc", + purpose: "adapter", responsibilities: ["argv-construction", "output-interpretation", "session-handling", "token-accounting"], upstreamIssues: [ "https://github.com/monad-developers/ultrafuzz/issues/1006", @@ -151,156 +122,11 @@ const responsibilityPolicies: Record = { "https://github.com/smithersai/smithers/issues/1629" ] }, - "provider-home.tsx": { - classifiedSourceSha256: "31085a2bad1d6d82b3709946464df332fe1d22e13236708dfb840c8fbd7d5744", - responsibilities: [], - upstreamIssues: [] - }, - "strict-json.tsx": { - classifiedSourceSha256: "16c909eb1f01c82e1174db61877a30028b58a49466714865f1243293f10b186b", - responsibilities: [], - upstreamIssues: [] - }, - "toml.tsx": { - classifiedSourceSha256: "51b15d0f75a09b49a53a33709cf9127c74b2a35814ad8770ce529f0638a4f6ae", - responsibilities: [], - upstreamIssues: [] - } -}; - -// Syntax-node ceilings and exact source fingerprints are the reviewed PR shape -// rooted at main@fe0922ea. The fingerprint makes every replacement visible -// even when it preserves or reduces aggregate structure; line ceilings retain -// a small formatting/documentation margin. -const sourcePolicies: Record = { - "claude.tsx": { - maxLines: 100, - maxSyntaxNodes: 452, - purpose: "adapter", - sourceSha256: "6d8346e882f9ed1e5274d19d6e7cacf27e2743e489a3b931b9ee0488d62ffe76" - }, - "codex.tsx": { - maxLines: 250, - maxSyntaxNodes: 1_300, - purpose: "adapter", - sourceSha256: "a44eb5c47e6374457476a86420eca0c5ff23616fc8201637f0139d37e13b92f5" - }, - "deepseek.tsx": { - // Raised with the 0.35.0 pin bump: the pinned BaseCliAgent now rejects - // unknown constructor options, so the adapter carries a thin constructor - // that splits the Ultrafuzz-only credential off `this.opts`. - maxLines: 360, - maxSyntaxNodes: 1_675, - purpose: "adapter", - sourceSha256: "ea7c6eec70883126e3ee9588b6d3349169652688756b519f1d49c3b5b794e887" - }, - "environment.tsx": { - // Shared native-continuation PATH and Pi home filtering (#1035). - maxLines: 425, - maxSyntaxNodes: 2_325, - purpose: "data-governance", - sourceSha256: "72d93b360e5b969e00646cfa39a24cac8a3720ce2aef1c7469c51038ef0206cd" - }, - "index.tsx": { - maxLines: 30, - maxSyntaxNodes: 125, - purpose: "registry", - sourceSha256: "ce5f94b3bf12ae40c5b59ebd587a77d1e80e532d92c785f3353d272e627d79e4" - }, - "kimi.tsx": { - // Raised with the 0.35.0 pin bump, for the same reason as deepseek.tsx. - maxLines: 1_600, - maxSyntaxNodes: 9_250, - purpose: "adapter", - sourceSha256: "1551f53080570f026aa8f6f60d533ef51abe8bac453eb0610e0ef11a9f974530" - }, - "opencode.tsx": { - maxLines: 150, - maxSyntaxNodes: 650, - purpose: "adapter", - sourceSha256: "7d4e22b674e06b00cd537c531c0e7e95d800ffc07b50d02b566a5fd029d0b489" - }, - "openrouter.tsx": { - // Raised after the accounting review for cumulative response aggregation, - // retry/failure usage, and adapter-recorded cost preservation (#1006). - maxLines: 1_750, - maxSyntaxNodes: 9_075, - purpose: "adapter", - sourceSha256: "834b8d1893f1a6b7da9667cd5f55ecd7fd5e7dc9179ab2b307562a9913f410b4" - }, - "pi.tsx": { - // Raised for cumulative per-response usage, session-aware progress, and - // adapter-recorded cost preservation in addition to terminal mapping. - maxLines: 475, - maxSyntaxNodes: 2_625, - purpose: "adapter", - sourceSha256: "1b9e81f7d79a7c7f551c1b6d7d1f794e2d49ee698b328eee4a872b914bfc8acc" - }, - "provider-home.tsx": { - maxLines: 75, - maxSyntaxNodes: 556, - purpose: "provider-home", - sourceSha256: "31085a2bad1d6d82b3709946464df332fe1d22e13236708dfb840c8fbd7d5744" - }, - "strict-json.tsx": { - maxLines: 350, - maxSyntaxNodes: 1_965, - purpose: "strict-input", - sourceSha256: "16c909eb1f01c82e1174db61877a30028b58a49466714865f1243293f10b186b" - }, - "toml.tsx": { - maxLines: 120, - maxSyntaxNodes: 526, - purpose: "toml", - sourceSha256: "51b15d0f75a09b49a53a33709cf9127c74b2a35814ad8770ce529f0638a4f6ae" - } + "provider-home.tsx": { purpose: "provider-home", responsibilities: [], upstreamIssues: [] }, + "strict-json.tsx": { purpose: "strict-input", responsibilities: [], upstreamIssues: [] }, + "toml.tsx": { purpose: "toml", responsibilities: [], upstreamIssues: [] } }; -function lineCount(source: string): number { - return source.replace(/\n$/u, "").split("\n").length; -} - -function syntaxNodeCount(sourceFile: ts.SourceFile): number { - let count = 0; - const visit = (node: ts.Node): void => { - if (node !== sourceFile) count += 1; - ts.forEachChild(node, visit); - }; - visit(sourceFile); - return count; -} - -function sourceFingerprint(source: string): string { - return crypto.createHash("sha256").update(source, "utf8").digest("hex"); -} - -function assertSourceMatchesPolicy(relativePath: string, source: SourceUnit, policy: SourcePolicy): void { - const lines = lineCount(source.source); - const syntaxNodes = syntaxNodeCount(source.ast); - assert.ok(lines <= policy.maxLines, `${relativePath} grew past its ${policy.maxLines}-line review ceiling`); - assert.ok( - syntaxNodes <= policy.maxSyntaxNodes, - `${relativePath} grew past its ${policy.maxSyntaxNodes}-node structural ceiling; classify the change before accepting it` - ); - assert.equal( - sourceFingerprint(source.source), - policy.sourceSha256, - `${relativePath} changed from its reviewed source fingerprint; audit responsibilities and update the policy explicitly` - ); -} - -function assertResponsibilityReviewMatchesSource( - relativePath: string, - source: SourceUnit, - policy: ResponsibilityPolicy -): void { - assert.equal( - sourceFingerprint(source.source), - policy.classifiedSourceSha256, - `${relativePath} changed from its responsibility-reviewed source fingerprint; audit its responsibility declaration and refresh the independent classification policy` - ); -} - function detectedOrchestratorResponsibilities(source: SourceUnit): ReadonlySet { const detected = new Set(); const filesystemWalkingBindings = new Set(); @@ -443,7 +269,7 @@ function detectedOrchestratorResponsibilities(source: SourceUnit): ReadonlySet ): readonly OrchestratorResponsibility[] { const detected = [...detectedOrchestratorResponsibilities(source)].sort(); const undeclared = detected.filter((responsibility) => !policy.responsibilities.includes(responsibility)); @@ -881,79 +707,7 @@ test("source inventory recurses through both TypeScript source extensions", () = } }); -test("syntax budgets ignore names and comments but catch structural orchestration additions", () => { - const baseline = syntaxNodeCount( - parseSource("fixture.ts", "export async function run(task: string) { return task; }\n").ast - ); - const renamed = syntaxNodeCount( - parseSource( - "fixture.ts", - "// resumeSession and prompt_tokens are documentation, not classification markers.\n" + - "export async function invoke(prompt: string) { return prompt; }\n" - ).ast - ); - assert.equal(renamed, baseline); - assert.notEqual( - sourceFingerprint("export async function run(task: string) { return task; }\n"), - sourceFingerprint( - "// resumeSession and prompt_tokens are documentation, not classification markers.\n" + - "export async function invoke(prompt: string) { return prompt; }\n" - ), - "an equal-size replacement must still require an explicit source-policy review" - ); - - for (const addition of [ - 'import { readdir } from "node:fs/promises"; export async function run() { return readdir("."); }\n', - 'export function run(prompt: string) { const launchArguments = ["--resume", prompt]; return launchArguments; }\n', - "export function run(usage: { prompt_tokens: number; completion_tokens: number }) { return usage.prompt_tokens + usage.completion_tokens; }\n" - ]) { - assert.ok(syntaxNodeCount(parseSource("fixture.ts", addition).ast) > baseline); - } -}); - -test("source-policy acknowledgment cannot reuse a stale responsibility review for opaque behavior", () => { - const baselineSource = "export function passThrough(payload: Uint8Array) { return payload; }\n"; - const changedSource = - 'import { decodeProviderEvent } from "./provider-parser";\n' + - "export function passThrough(payload: Uint8Array) { return decodeProviderEvent(payload); }\n"; - const changed = parseSource("fixture.tsx", changedSource); - const acknowledgedSourcePolicy: SourcePolicy = { - maxLines: lineCount(changedSource), - maxSyntaxNodes: syntaxNodeCount(changed.ast), - purpose: "adapter", - sourceSha256: sourceFingerprint(changedSource) - }; - const staleResponsibilityPolicy: ResponsibilityPolicy = { - classifiedSourceSha256: sourceFingerprint(baselineSource), - responsibilities: [], - upstreamIssues: [] - }; - - assert.deepEqual( - [...detectedOrchestratorResponsibilities(changed)], - [], - "the fixture must exercise a semantic form outside the conservative static lower bound" - ); - assert.doesNotThrow(() => assertSourceMatchesPolicy("fixture.tsx", changed, acknowledgedSourcePolicy)); - assert.throws( - () => assertResponsibilityReviewMatchesSource("fixture.tsx", changed, staleResponsibilityPolicy), - /responsibility-reviewed source fingerprint/u - ); - - const refreshedResponsibilityPolicy: ResponsibilityPolicy = { - classifiedSourceSha256: sourceFingerprint(changedSource), - responsibilities: ["output-interpretation"], - upstreamIssues: ["https://example.invalid/upstream"] - }; - assert.doesNotThrow(() => - assertResponsibilityReviewMatchesSource("fixture.tsx", changed, refreshedResponsibilityPolicy) - ); - assert.doesNotThrow(() => - assertDetectedResponsibilitiesDeclared("fixture.tsx", changed, refreshedResponsibilityPolicy) - ); -}); - -test("fingerprint and ceiling acknowledgment cannot retain stale responsibility classifications", () => { +test("a detected responsibility fails until the source declares it", () => { const changedSources: Array<[OrchestratorResponsibility, string]> = [ [ "filesystem-walking", @@ -980,33 +734,13 @@ test("fingerprint and ceiling acknowledgment cannot retain stale responsibility for (const [responsibility, changedSource] of changedSources) { const parsed = parseSource("fixture.tsx", changedSource); - const acknowledgedSourcePolicy: SourcePolicy = { - maxLines: lineCount(changedSource), - maxSyntaxNodes: syntaxNodeCount(parsed.ast), - purpose: "adapter", - sourceSha256: sourceFingerprint(changedSource) - }; - assert.doesNotThrow( - () => assertSourceMatchesPolicy("fixture.tsx", parsed, acknowledgedSourcePolicy), - `${responsibility} fixture must model an independently acknowledged fingerprint and ceiling` - ); - const staleCentralPolicy: ResponsibilityPolicy = { - classifiedSourceSha256: sourceFingerprint(changedSource), - responsibilities: [], - upstreamIssues: [] - }; - assert.doesNotThrow(() => assertResponsibilityReviewMatchesSource("fixture.tsx", parsed, staleCentralPolicy)); assert.throws( - () => assertDetectedResponsibilitiesDeclared("fixture.tsx", parsed, staleCentralPolicy), + () => assertDetectedResponsibilitiesDeclared("fixture.tsx", parsed, { responsibilities: [] }), /static signals for undeclared orchestrator responsibilities/u, responsibility ); assert.doesNotThrow(() => - assertDetectedResponsibilitiesDeclared("fixture.tsx", parsed, { - classifiedSourceSha256: sourceFingerprint(changedSource), - responsibilities: [responsibility], - upstreamIssues: ["https://example.invalid/upstream"] - }) + assertDetectedResponsibilitiesDeclared("fixture.tsx", parsed, { responsibilities: [responsibility] }) ); } }); @@ -1095,7 +829,7 @@ test("non-adapter helpers cannot hide orchestrator responsibilities", () => { }); test("OpenRouter retains the manually reviewed argv responsibility inherited from Codex", () => { - assert.equal(responsibilityPolicies["openrouter.tsx"]!.responsibilities.includes("argv-construction"), true); + assert.equal(adapterPolicies["openrouter.tsx"]?.responsibilities.includes("argv-construction"), true); }); test("main agent registry and recursive sources stay inside reviewed adapter boundaries", (context) => { @@ -1106,20 +840,15 @@ test("main agent registry and recursive sources stay inside reviewed adapter bou const sources = readSourceTree(path.join(packageRoot, "src/templates/smithers/agents")); assert.deepEqual( - Object.keys(sourcePolicies).sort(), + Object.keys(adapterPolicies).sort(), [...sources.keys()].sort(), - "every recursive .ts/.tsx adapter source must have an explicit structural policy" - ); - assert.deepEqual( - Object.keys(responsibilityPolicies).sort(), - [...sources.keys()].sort(), - "every recursive .ts/.tsx source must have an independent responsibility-review policy" + "every recursive .ts/.tsx adapter source must declare its purpose and responsibilities" ); const registered = registeredAdapterSources(sources); const registeredSourcePaths = [...new Set(registered.values())].sort(); assert.deepEqual( - Object.entries(sourcePolicies) + Object.entries(adapterPolicies) .filter(([, policy]) => policy.purpose === "adapter") .map(([relativePath]) => relativePath) .sort(), @@ -1127,23 +856,16 @@ test("main agent registry and recursive sources stay inside reviewed adapter bou "only adapter sources registered in agentFactories may carry the adapter purpose" ); for (const [relativePath, source] of sources) { - const policy = sourcePolicies[relativePath]!; - const responsibilityPolicy = responsibilityPolicies[relativePath]!; - const lines = lineCount(source.source); - const syntaxNodes = syntaxNodeCount(source.ast); - context.diagnostic( - `${relativePath}: ${lines} lines; ${syntaxNodes} syntax nodes; reviewed purpose: ${policy.purpose}` - ); - assertSourceMatchesPolicy(relativePath, source, policy); - assertResponsibilityReviewMatchesSource(relativePath, source, responsibilityPolicy); + const policy = adapterPolicies[relativePath]; + assert.ok(policy, `${relativePath} has no adapter policy`); if (policy.purpose !== "adapter") { assert.deepEqual( - responsibilityPolicy.responsibilities, + policy.responsibilities, [], `${relativePath} is not a registered adapter and cannot own orchestrator responsibilities` ); assert.deepEqual( - responsibilityPolicy.upstreamIssues, + policy.upstreamIssues, [], `${relativePath} is not a registered adapter and cannot own adapter debt` ); @@ -1152,12 +874,10 @@ test("main agent registry and recursive sources stay inside reviewed adapter bou } for (const adapterSource of registeredSourcePaths) { - const policy = responsibilityPolicies[adapterSource]!; - const detectedResponsibilities = assertDetectedResponsibilitiesDeclared( - adapterSource, - sources.get(adapterSource)!, - policy - ); + const policy = adapterPolicies[adapterSource]; + const source = sources.get(adapterSource); + assert.ok(policy && source, `${adapterSource} has no adapter policy`); + const detectedResponsibilities = assertDetectedResponsibilitiesDeclared(adapterSource, source, policy); const registrations = [...registered] .filter(([, sourcePath]) => sourcePath === adapterSource) .map(([agentRef]) => agentRef) @@ -1167,7 +887,6 @@ test("main agent registry and recursive sources stay inside reviewed adapter bou policy.responsibilities.join(", ") || "none" }; statically detected lower bound: ${detectedResponsibilities.join(", ") || "none"}` ); - assert.equal(sourcePolicies[adapterSource]?.purpose, "adapter"); if (policy.responsibilities.length > 0) { assert.ok(policy.upstreamIssues.length > 0, `${adapterSource} debt must link an upstream issue`); } diff --git a/packages/runtime/test/aggregation-semantic-context.test.ts b/packages/runtime/test/aggregation-semantic-context.test.ts deleted file mode 100644 index ca39e1976..000000000 --- a/packages/runtime/test/aggregation-semantic-context.test.ts +++ /dev/null @@ -1,725 +0,0 @@ -import assert from "node:assert/strict"; -import { temporaryRoot } from "./temporary-root.js"; -import crypto from "node:crypto"; -import fs from "node:fs"; -import path from "node:path"; -import test from "node:test"; - -import { - ARTIFACT_VERIFICATION_SCHEMA_VERSION, - PLANNED_GRAPH_SCHEMA_VERSION, - artifactContractDefinition, - artifactContractSchemaBinding, - createRunLayout, - executeSchemaSemanticGates, - writeArtifact, - writeArtifactManifest, - writeJsonDurable, - type ArtifactContractId, - type ArtifactManifest, - type ArtifactVerificationMarker, - type PlannedGraphDocument, - type PlannedGraphNodeDocument, - type PlannedGraphOutput, - type RunLayout -} from "@ultrafuzz/artifacts"; - -import { authenticatedAggregationSemanticContext } from "../src/aggregation-semantic-context.js"; - -interface AggregationAuthorityFixture { - layout: RunLayout; - aggregationNode: PlannedGraphNodeDocument; - expectedPrerequisiteAttemptIds: string[]; - prerequisiteManifestPaths: string[]; - originManifestPath: string; - unrelatedManifestPath: string; - markerPath: string; - artifactManifestPath: string; - generatedManifestPath: string; - generatedTestPath: string; - supportFilePath: string; -} - -interface RunFileSnapshot { - bytes: Buffer; - dev: bigint; - ino: bigint; - nlink: bigint; - size: bigint; - mtimeNs: bigint; - ctimeNs: bigint; -} - -interface RunTreeSnapshot { - directories: string[]; - files: Map; -} - -test("authenticated aggregation context accepts the exact sealed producer authority", (t) => { - const fixture = createAggregationAuthorityFixture("aggregation-authority-valid"); - t.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - - const before = snapshotRunTree(fixture.layout.root); - const context = authenticatedContext(fixture); - - assert.equal(context.sourceBundles.length, 1); - assert.deepEqual( - context.sourceBundles.map((bundle) => ({ - strategy: bundle.strategy, - sourceAttemptId: bundle.sourceAttemptId, - sourceManifestRelativePath: bundle.sourceManifestRelativePath, - sourceRunId: bundle.sourceRunId, - framework: bundle.framework, - entries: bundle.entries.map((entry) => ({ kind: entry.kind, path: entry.sourceRelativePath })) - })), - [ - { - strategy: "generated-producer", - sourceAttemptId: "generated-producer", - sourceManifestRelativePath: "generated-tests.json", - sourceRunId: "aggregation-authority-valid", - framework: "foundry", - entries: [ - { kind: "generated-test", path: "generated-tests/Property.t.sol" }, - { kind: "support-file", path: "generated-tests/PropertyHelper.sol" } - ] - } - ] - ); - assert.deepEqual(snapshotRunTree(fixture.layout.root), before); -}); - -test("authenticated aggregation source authority rejects hard-linked sealed inputs without mutation", async (t) => { - const cases: Array<{ name: string; select: (fixture: AggregationAuthorityFixture) => string }> = [ - { name: "verification marker", select: (fixture) => fixture.markerPath }, - { name: "artifact manifest", select: (fixture) => fixture.artifactManifestPath }, - { name: "generated-test manifest", select: (fixture) => fixture.generatedManifestPath }, - { name: "declared companion", select: (fixture) => fixture.supportFilePath } - ]; - - for (const [index, attack] of cases.entries()) { - await t.test(attack.name, (subtest) => { - const fixture = createAggregationAuthorityFixture(`aggregation-hardlink-${index}`); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - const target = attack.select(fixture); - const attackerDirectory = path.join(fixture.layout.root, "attacker-links"); - fs.mkdirSync(attackerDirectory); - fs.linkSync(target, path.join(attackerDirectory, `${index}.alias`)); - assert.equal(fs.lstatSync(target, { bigint: true }).nlink, 2n); - - assertAuthorityRejectedWithoutMutation(fixture, /must be a singly linked regular file/u); - assert.equal(fs.lstatSync(target, { bigint: true }).nlink, 2n); - }); - } -}); - -test("authenticated aggregation source authority requires exact marker and artifact-manifest publication sets", async (t) => { - const attacks: Array<{ - name: string; - mutate: (fixture: AggregationAuthorityFixture) => void; - }> = [ - { - name: "marker omits a declared companion", - mutate(fixture) { - mutateMarker(fixture, (marker) => { - marker.publications = marker.publications.filter( - (publication) => publication.path !== "generated-tests/PropertyHelper.sol" - ); - }); - } - }, - { - name: "marker fabricates an extra publication", - mutate(fixture) { - mutateMarker(fixture, (marker) => { - marker.publications.push({ path: "generated-tests/Extra.sol", sha256: "0".repeat(64) }); - }); - } - }, - { - name: "artifact manifest omits a declared companion", - mutate(fixture) { - mutateArtifactManifest(fixture, (manifest) => { - manifest.files = manifest.files.filter((entry) => entry.path !== "generated-tests/PropertyHelper.sol"); - }); - } - }, - { - name: "artifact manifest fabricates an extra publication", - mutate(fixture) { - mutateArtifactManifest(fixture, (manifest) => { - const template = manifest.files.find((entry) => entry.path === "generated-tests/PropertyHelper.sol"); - assert.ok(template); - manifest.files.push({ ...structuredClone(template), path: "generated-tests/Extra.sol" }); - }); - } - } - ]; - - for (const [index, attack] of attacks.entries()) { - await t.test(attack.name, (subtest) => { - const fixture = createAggregationAuthorityFixture(`aggregation-publications-${index}`); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - attack.mutate(fixture); - assertAuthorityRejectedWithoutMutation(fixture, /publication sets differ/u); - }); - } -}); - -test("authenticated aggregation source authority rejects prerequisite fanout attempt-set drift", async (t) => { - const attacks: Array<{ - name: string; - mutate: (fixture: AggregationAuthorityFixture, manifest: ArtifactManifest) => void; - }> = [ - { - name: "one planned prerequisite attempt is omitted", - mutate(fixture, manifest) { - manifest.prerequisite_manifests = manifest.prerequisite_manifests.filter( - (entry) => entry.node_id !== fixture.expectedPrerequisiteAttemptIds[1] - ); - } - }, - { - name: "a planned prerequisite attempt is replaced with a fabricated attempt", - mutate(_fixture, manifest) { - manifest.prerequisite_manifests[1] = { - ...manifest.prerequisite_manifests[1]!, - node_id: "seed__model_9__attempt_9" - }; - } - } - ]; - - for (const [index, attack] of attacks.entries()) { - await t.test(attack.name, (subtest) => { - const fixture = createAggregationAuthorityFixture(`aggregation-prerequisites-${index}`); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - assert.deepEqual( - readJsonFile(fixture.artifactManifestPath).prerequisite_manifests.map( - (entry) => entry.node_id - ), - fixture.expectedPrerequisiteAttemptIds - ); - mutateArtifactManifest(fixture, (manifest) => attack.mutate(fixture, manifest)); - assertAuthorityRejectedWithoutMutation(fixture, /prerequisite set changed/u); - }); - } -}); - -test("authenticated aggregation source authority snapshots the sealed prerequisite manifest chain", async (t) => { - await t.test("hard-linked prerequisite manifest", (subtest) => { - const fixture = createAggregationAuthorityFixture("aggregation-prerequisite-hardlink"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - const manifestPath = fixture.prerequisiteManifestPaths[0]!; - const aliasPath = path.join(fixture.layout.root, "prerequisite-manifest.alias.json"); - fs.linkSync(manifestPath, aliasPath); - - assertAuthorityRejectedWithoutMutation(fixture, /must be a singly linked regular file/u); - assert.equal(fs.lstatSync(manifestPath).nlink, 2); - assert.deepEqual(fs.readFileSync(aliasPath), fs.readFileSync(manifestPath)); - }); - - await t.test("stale prerequisite manifest digest", (subtest) => { - const fixture = createAggregationAuthorityFixture("aggregation-prerequisite-byte-drift"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - const manifestPath = fixture.prerequisiteManifestPaths[0]!; - const manifest = readJsonFile(manifestPath); - manifest.created_at = "2026-08-10T00:00:00.000Z"; - writeJsonForAttack(manifestPath, manifest); - - assertAuthorityRejectedWithoutMutation(fixture, /prerequisite artifact manifest bytes changed/u); - }); -}); - -test("authenticated aggregation source authority reconciles every transitive manifest with the planned graph", async (t) => { - await t.test("stale grandparent bytes", (subtest) => { - const fixture = createAggregationAuthorityFixture("aggregation-stale-grandparent"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - mutateManifestAtPath(fixture.originManifestPath, (manifest) => { - manifest.created_at = "2026-08-10T00:00:01.000Z"; - }); - - assertAuthorityRejectedWithoutMutation(fixture, /prerequisite artifact manifest bytes changed for origin/u); - }); - - await t.test("omitted planned grandparent after attacker reseals descendants", (subtest) => { - const fixture = createAggregationAuthorityFixture("aggregation-omitted-grandparent"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - mutateManifestAtPath(fixture.prerequisiteManifestPaths[0]!, (manifest) => { - manifest.prerequisite_manifests = []; - }); - resealProducerPrerequisiteDigests(fixture); - - assertAuthorityRejectedWithoutMutation( - fixture, - /prerequisite artifact manifest prerequisite set changed for seed__model_0__attempt_0/u - ); - }); - - await t.test("fabricated graph-known nondependency after attacker reseals descendants", (subtest) => { - const fixture = createAggregationAuthorityFixture("aggregation-fabricated-grandparent"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - mutateManifestAtPath(fixture.prerequisiteManifestPaths[0]!, (manifest) => { - manifest.prerequisite_manifests.push({ - node_id: "unrelated-generated-producer", - sha256: digest(fs.readFileSync(fixture.unrelatedManifestPath)) - }); - }); - resealProducerPrerequisiteDigests(fixture); - - assertAuthorityRejectedWithoutMutation( - fixture, - /prerequisite artifact manifest prerequisite set changed for seed__model_0__attempt_0/u - ); - }); - - await t.test("transitive manifest identity drift after attacker reseals the complete chain", (subtest) => { - const fixture = createAggregationAuthorityFixture("aggregation-grandparent-identity"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - mutateManifestAtPath(fixture.originManifestPath, (manifest) => { - manifest.provenance.logical_node_id = "unrelated-generated-producer"; - }); - const originSha256 = digest(fs.readFileSync(fixture.originManifestPath)); - for (const manifestPath of fixture.prerequisiteManifestPaths) { - mutateManifestAtPath(manifestPath, (manifest) => { - manifest.prerequisite_manifests[0]!.sha256 = originSha256; - }); - } - resealProducerPrerequisiteDigests(fixture); - - assertAuthorityRejectedWithoutMutation(fixture, /prerequisite artifact manifest identity changed for origin/u); - }); - - await t.test("diamond ancestors with conflicting sealed digests", (subtest) => { - const fixture = createAggregationAuthorityFixture("aggregation-conflicting-diamond"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - mutateManifestAtPath(fixture.prerequisiteManifestPaths[0]!, (manifest) => { - manifest.prerequisite_manifests[0]!.sha256 = "f".repeat(64); - }); - resealProducerPrerequisiteDigests(fixture); - - assertAuthorityRejectedWithoutMutation( - fixture, - /prerequisite artifact manifest has conflicting sealed digests for origin/u - ); - }); -}); - -test("authenticated aggregation context excludes graph-known generated-test nondependencies", (t) => { - const fixture = createAggregationAuthorityFixture("aggregation-nondependency-producer"); - t.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - - const context = authenticatedContext(fixture); - - assert.deepEqual( - context.sourceBundles.map((bundle) => bundle.sourceAttemptId), - ["generated-producer"] - ); - assert.equal( - fs.existsSync(path.join(fixture.layout.root, ".ultrafuzz-verification", "unrelated-generated-producer.json")), - false - ); -}); - -test("authenticated aggregation context distinguishes valid empty authority from a missing required producer", async (t) => { - await t.test("one authenticated empty bundle remains an explicit typed source bundle", (subtest) => { - const fixture = createAggregationAuthorityFixture("aggregation-empty-bundle"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - makeGeneratedProducerBundleEmpty(fixture); - const before = snapshotRunTree(fixture.layout.root); - - const context = authenticatedContext(fixture); - - assert.equal(context.sourceBundles.length, 1); - assert.equal(context.sourceBundles[0]!.framework, "foundry"); - assert.deepEqual(context.sourceBundles[0]!.entries, []); - assert.deepEqual(snapshotRunTree(fixture.layout.root), before); - }); - - await t.test("an exact graph with no generated-test producers admits the canonical empty aggregation", (subtest) => { - const fixture = createNoGeneratedProducerFixture("aggregation-empty-producer-set"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - const before = snapshotRunTree(fixture.layout.root); - - const context = authenticatedAggregationSemanticContext({ - layout: fixture.layout, - node: fixture.aggregationNode, - attemptId: fixture.aggregationNode.id - }); - const document = { - schema_version: "ultrafuzz.aggregation-manifest.v1", - source_generated_tests: 0, - copied_generated_tests: 0, - source_support_files: 0, - copied_support_files: 0, - source_bundles: [], - files: [], - support_files: [], - skipped_files: [] - }; - const results = executeSchemaSemanticGates("aggregation-manifest.schema.json", { - document, - context: { aggregation: context } - }); - - assert.deepEqual(context.sourceBundles, []); - assert.deepEqual( - results.filter((result) => result.status !== "passed"), - [] - ); - assert.deepEqual(snapshotRunTree(fixture.layout.root), before); - }); - - await t.test("a declared generated-test producer without verifier authority fails closed", (subtest) => { - const fixture = createAggregationAuthorityFixture("aggregation-missing-producer-marker"); - subtest.after(() => fs.rmSync(path.dirname(fixture.layout.root), { recursive: true, force: true })); - fs.rmSync(fixture.markerPath); - - assertAuthorityRejectedWithoutMutation(fixture, /artifact verification marker/u); - }); -}); - -function createNoGeneratedProducerFixture(runId: string): { - layout: RunLayout; - aggregationNode: PlannedGraphNodeDocument; -} { - const outputRoot = temporaryRoot("ultrafuzz-empty-aggregation-authority-"); - const seedNode = plannedNode("seed", [], boundOutput("seed.md", "ultrafuzz/nonempty-markdown@1")); - const aggregationNode = plannedNode( - "aggregate-test-files", - [seedNode.id], - boundOutput("aggregation-manifest.json", "ultrafuzz/aggregation-manifest@1") - ); - const graph: PlannedGraphDocument = { - schema_version: PLANNED_GRAPH_SCHEMA_VERSION, - graph_version: "4", - topology_version: 2, - groups: {}, - nodes: [seedNode, aggregationNode] - }; - const layout = createRunLayout({ outputRoot, runId, graph }); - fs.mkdirSync(path.join(layout.workspacesDir, aggregationNode.id), { recursive: true }); - return { layout, aggregationNode }; -} - -function makeGeneratedProducerBundleEmpty(fixture: AggregationAuthorityFixture): void { - const generatedManifest = readJsonFile<{ - generated_tests: unknown[]; - support_files: unknown[]; - }>(fixture.generatedManifestPath); - generatedManifest.generated_tests = []; - generatedManifest.support_files = []; - writeJsonForAttack(fixture.generatedManifestPath, generatedManifest); - fs.rmSync(fixture.generatedTestPath); - fs.rmSync(fixture.supportFilePath); - - const manifestSha256 = digest(fs.readFileSync(fixture.generatedManifestPath)); - const manifestSize = fs.statSync(fixture.generatedManifestPath).size; - mutateArtifactManifest(fixture, (manifest) => { - manifest.files = manifest.files.filter( - (entry) => entry.path !== "generated-tests/Property.t.sol" && entry.path !== "generated-tests/PropertyHelper.sol" - ); - const generatedManifestEntry = manifest.files.find((entry) => entry.path === "generated-tests.json"); - assert.ok(generatedManifestEntry); - generatedManifestEntry.sha256 = manifestSha256; - generatedManifestEntry.size_bytes = manifestSize; - }); - mutateMarker(fixture, (marker) => { - marker.publications = marker.publications.filter( - (entry) => entry.path !== "generated-tests/Property.t.sol" && entry.path !== "generated-tests/PropertyHelper.sol" - ); - const generatedArtifact = marker.artifacts.find((entry) => entry.path === "generated-tests.json"); - const generatedPublication = marker.publications.find((entry) => entry.path === "generated-tests.json"); - assert.ok(generatedArtifact); - assert.ok(generatedPublication); - generatedArtifact.sha256 = manifestSha256; - generatedPublication.sha256 = manifestSha256; - }); -} - -function resealProducerPrerequisiteDigests(fixture: AggregationAuthorityFixture): void { - const currentDigests = new Map( - fixture.prerequisiteManifestPaths.map((manifestPath) => [ - path.basename(path.dirname(manifestPath)), - digest(fs.readFileSync(manifestPath)) - ]) - ); - mutateArtifactManifest(fixture, (manifest) => { - for (const prerequisite of manifest.prerequisite_manifests) { - const sha256 = currentDigests.get(prerequisite.node_id); - if (sha256 !== undefined) prerequisite.sha256 = sha256; - } - }); -} - -function createAggregationAuthorityFixture(runId: string): AggregationAuthorityFixture { - const outputRoot = temporaryRoot("ultrafuzz-aggregation-authority-"); - const originOutput = boundOutput("origin.md", "ultrafuzz/nonempty-markdown@1"); - const seedOutput = boundOutput("seed.md", "ultrafuzz/nonempty-markdown@1"); - const generatedOutput = boundOutput("generated-tests.json", "ultrafuzz/generated-tests@3"); - const aggregationOutput = boundOutput("aggregation-manifest.json", "ultrafuzz/aggregation-manifest@1"); - const originNode = plannedNode("origin", [], originOutput); - const seedNode: PlannedGraphNodeDocument = { - ...plannedNode("seed", [originNode.id], seedOutput), - model_fanout: [ - { - model_profile_id: "model-zero", - agent_ref: "CodexAgent", - model_index: 0, - loop_index: 0, - attempt_index: 0 - }, - { - model_profile_id: "model-one", - agent_ref: "CodexAgent", - model_index: 1, - loop_index: 0, - attempt_index: 0 - } - ] - }; - const producerNode = plannedNode("generated-producer", [seedNode.id], generatedOutput); - const unrelatedNode = plannedNode("unrelated-generated-producer", [], generatedOutput); - const aggregationNode = plannedNode("aggregate-test-files", [producerNode.id], aggregationOutput); - const graph: PlannedGraphDocument = { - schema_version: PLANNED_GRAPH_SCHEMA_VERSION, - graph_version: "4", - topology_version: 2, - groups: {}, - nodes: [originNode, seedNode, producerNode, unrelatedNode, aggregationNode] - }; - const layout = createRunLayout({ outputRoot, runId, graph }); - writeArtifact(layout, originNode.id, originOutput.path, "origin\n"); - writeArtifactManifest({ - layout, - nodeId: originNode.id, - outputs: [originOutput], - provenance: { - logical_node_id: originNode.logical_id, - attempt_index: 0, - metadata: { concrete_node_id: originNode.id } - } - }); - const originManifestPath = path.join(layout.artifactsDir, originNode.id, "artifact-manifest.json"); - - writeArtifact( - layout, - unrelatedNode.id, - generatedOutput.path, - `${JSON.stringify({ - schema_version: "ultrafuzz.generated-tests.v3", - run_id: runId, - node_id: unrelatedNode.logical_id, - framework: "medusa", - generated_tests: [], - support_files: [] - })}\n` - ); - writeArtifactManifest({ - layout, - nodeId: unrelatedNode.id, - outputs: [generatedOutput], - provenance: { - logical_node_id: unrelatedNode.logical_id, - attempt_index: 0, - metadata: { concrete_node_id: unrelatedNode.id } - } - }); - const unrelatedManifestPath = path.join(layout.artifactsDir, unrelatedNode.id, "artifact-manifest.json"); - - const expectedPrerequisiteAttemptIds = ["seed__model_0__attempt_0", "seed__model_1__attempt_0"]; - const prerequisiteManifestPaths: string[] = []; - for (const [modelIndex, attemptId] of expectedPrerequisiteAttemptIds.entries()) { - writeArtifact(layout, attemptId, seedOutput.path, `seed ${modelIndex}\n`); - writeArtifactManifest({ - layout, - nodeId: attemptId, - outputs: [seedOutput], - prerequisiteNodeIds: [originNode.id], - provenance: { - logical_node_id: seedNode.logical_id, - attempt_index: 0, - model_index: modelIndex, - metadata: { concrete_node_id: seedNode.id } - } - }); - prerequisiteManifestPaths.push(path.join(layout.artifactsDir, attemptId, "artifact-manifest.json")); - } - - const generatedTestContents = Buffer.from("contract Property {}\n", "utf8"); - const supportFileContents = Buffer.from("library PropertyHelper {}\n", "utf8"); - const generatedTestRelativePath = "generated-tests/Property.t.sol"; - const supportFileRelativePath = "generated-tests/PropertyHelper.sol"; - const generatedManifest = { - schema_version: "ultrafuzz.generated-tests.v3", - run_id: runId, - node_id: producerNode.logical_id, - framework: "foundry", - generated_tests: [generatedTestEntry(generatedTestRelativePath, generatedTestContents)], - support_files: [generatedTestEntry(supportFileRelativePath, supportFileContents)] - }; - const generatedTestPath = writeArtifact(layout, producerNode.id, generatedTestRelativePath, generatedTestContents); - const supportFilePath = writeArtifact(layout, producerNode.id, supportFileRelativePath, supportFileContents); - const generatedManifestPath = writeArtifact( - layout, - producerNode.id, - generatedOutput.path, - `${JSON.stringify(generatedManifest)}\n` - ); - const artifactManifest = writeArtifactManifest({ - layout, - nodeId: producerNode.id, - outputs: [generatedOutput], - prerequisiteNodeIds: expectedPrerequisiteAttemptIds, - provenance: { - logical_node_id: producerNode.logical_id, - attempt_index: 0, - metadata: { concrete_node_id: producerNode.id } - } - }); - const generatedManifestPublication = artifactManifest.files.find((entry) => entry.path === generatedOutput.path); - assert.ok(generatedManifestPublication); - const marker: ArtifactVerificationMarker = { - schema_version: ARTIFACT_VERIFICATION_SCHEMA_VERSION, - attempt_id: producerNode.id, - node_id: producerNode.logical_id, - artifacts: [{ ...generatedOutput, sha256: generatedManifestPublication.sha256 }], - publications: artifactManifest.files.map((entry) => ({ path: entry.path, sha256: entry.sha256 })) - }; - const markerPath = path.join(layout.root, ".ultrafuzz-verification", `${producerNode.id}.json`); - writeJsonDurable(markerPath, marker); - - return { - layout, - aggregationNode, - expectedPrerequisiteAttemptIds, - prerequisiteManifestPaths, - originManifestPath, - unrelatedManifestPath, - markerPath, - artifactManifestPath: path.join(layout.artifactsDir, producerNode.id, "artifact-manifest.json"), - generatedManifestPath, - generatedTestPath, - supportFilePath - }; -} - -function plannedNode(id: string, dependsOn: string[], output: PlannedGraphOutput): PlannedGraphNodeDocument { - return { - id, - logical_id: id, - display_name: id, - kind: "agentic", - depends_on: dependsOn, - artifact_dir: `artifacts/${id}`, - outputs: [output], - prompt_id: id, - prompt_path: `prompts/${id}.md`, - loop: { index: 0, count: 1, mode: "parallel", attempt_index: 0 }, - model_fanout: [] - }; -} - -function boundOutput(artifactPath: string, contract: ArtifactContractId): PlannedGraphOutput { - return { - path: artifactPath, - contract, - contract_digest: artifactContractDefinition(contract).digest, - ...(artifactContractSchemaBinding(contract) ?? {}), - primary: true - }; -} - -function generatedTestEntry( - entryPath: string, - contents: Buffer -): { - path: string; - size_bytes: number; - sha256: string; -} { - return { - path: entryPath, - size_bytes: contents.byteLength, - sha256: digest(contents) - }; -} - -function authenticatedContext(fixture: AggregationAuthorityFixture) { - return authenticatedAggregationSemanticContext({ - layout: fixture.layout, - node: fixture.aggregationNode, - attemptId: fixture.aggregationNode.id - }); -} - -function assertAuthorityRejectedWithoutMutation(fixture: AggregationAuthorityFixture, expected: RegExp): void { - const before = snapshotRunTree(fixture.layout.root); - assert.throws(() => authenticatedContext(fixture), expected); - assert.deepEqual(snapshotRunTree(fixture.layout.root), before); -} - -function mutateMarker( - fixture: AggregationAuthorityFixture, - mutate: (marker: ArtifactVerificationMarker) => void -): void { - const marker = readJsonFile(fixture.markerPath); - mutate(marker); - writeJsonForAttack(fixture.markerPath, marker); -} - -function mutateArtifactManifest( - fixture: AggregationAuthorityFixture, - mutate: (manifest: ArtifactManifest) => void -): void { - mutateManifestAtPath(fixture.artifactManifestPath, mutate); -} - -function mutateManifestAtPath(filePath: string, mutate: (manifest: ArtifactManifest) => void): void { - const manifest = readJsonFile(filePath); - mutate(manifest); - writeJsonForAttack(filePath, manifest); -} - -function readJsonFile(filePath: string): T { - return JSON.parse(fs.readFileSync(filePath, "utf8")) as T; -} - -function writeJsonForAttack(filePath: string, value: unknown): void { - fs.writeFileSync(filePath, `${JSON.stringify(value)}\n`, "utf8"); -} - -function snapshotRunTree(root: string): RunTreeSnapshot { - const directories: string[] = []; - const files = new Map(); - const visit = (directory: string): void => { - for (const entry of fs - .readdirSync(directory, { withFileTypes: true }) - .sort((left, right) => left.name.localeCompare(right.name))) { - const absolutePath = path.join(directory, entry.name); - const relativePath = path.relative(root, absolutePath).split(path.sep).join("/"); - if (entry.isDirectory()) { - directories.push(relativePath); - visit(absolutePath); - continue; - } - const stat = fs.lstatSync(absolutePath, { bigint: true }); - assert.equal(stat.isFile(), true, `unexpected non-file in authority fixture: ${relativePath}`); - files.set(relativePath, { - bytes: fs.readFileSync(absolutePath), - dev: stat.dev, - ino: stat.ino, - nlink: stat.nlink, - size: stat.size, - mtimeNs: stat.mtimeNs, - ctimeNs: stat.ctimeNs - }); - } - }; - visit(root); - return { directories, files }; -} - -function digest(bytes: Uint8Array): string { - return crypto.createHash("sha256").update(bytes).digest("hex"); -} diff --git a/packages/runtime/test/artifact-gates.test.ts b/packages/runtime/test/artifact-gates.test.ts index 703a32d38..ca8bfaecf 100644 --- a/packages/runtime/test/artifact-gates.test.ts +++ b/packages/runtime/test/artifact-gates.test.ts @@ -543,8 +543,36 @@ function verifyRequiredArtifactsForAttempt( if (existingIndex === -1) graph.nodes.push(planned); else graph.nodes[existingIndex] = planned; fs.writeFileSync(layout.graphPath, JSON.stringify(graph), "utf8"); - writeSealedFixtureTaskAuthority(layout, graph.nodes, attemptAuthority?.tasks); - return verifyRuntimeRequiredArtifactsForAttempt(layout, planned, attemptId, attemptAuthority, authenticated); + const tasks = writeSealedFixtureTaskAuthority(layout, graph.nodes, attemptAuthority?.tasks); + return verifyRuntimeRequiredArtifactsForAttempt( + layout, + planned, + attemptId, + attemptAuthority ?? fixtureAttemptAuthority(tasks, attemptId), + authenticated + ); +} + +/** + * The sealed task set production passes, taken from the fixture's written task + * manifest. Verifier-admitted dependency attempt IDs are not modeled. + */ +function fixtureAttemptAuthority( + tasks: readonly SmithersTaskManifestTask[], + attemptId: string +): ArtifactGateAttemptAuthority { + const task = tasks.find((candidate) => candidate.attemptId === attemptId); + if (task === undefined) throw new Error(`fixture has no sealed task for attempt ${attemptId}`); + return { task, tasks }; +} + +/** Seal the planned graph exactly as written, without the wrapper's inferred dependencies. */ +function sealedFixtureAuthority( + layout: ReturnType, + attemptId: string +): ArtifactGateAttemptAuthority { + const graph = JSON.parse(fs.readFileSync(layout.graphPath, "utf8")) as { nodes: PlannedGraphNode[] }; + return fixtureAttemptAuthority(writeSealedFixtureTaskAuthority(layout, graph.nodes), attemptId); } function fixtureAttemptIds(node: PlannedGraphNode): string[] { @@ -615,9 +643,9 @@ function writeSealedFixtureTaskAuthority( layout: ReturnType, nodes: readonly PlannedGraphNode[], suppliedTasks?: readonly SmithersTaskManifestTask[] -): void { +): readonly SmithersTaskManifestTask[] { const sealedNodes = nodes.map((node) => { - if (node.kind !== "agentic" || node.workflow !== undefined) return node; + if (node.kind !== "agentic" || node.workflow !== undefined || node.dynamic !== undefined) return node; const attempts = fixtureAttemptIds(node); return { ...node, @@ -679,6 +707,7 @@ function writeSealedFixtureTaskAuthority( .sort() } }); + return tasks; } function boundOutput( @@ -2508,6 +2537,57 @@ test("severity classification gates preserve triaged fields and enforce the fina ); }); +test("severity classification keeps a false-positive record without severity, impact, or likelihood", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-severity-false-positive" }); + const triageNode: PlannedGraphNode = { + ...plannedNode([]), + id: "triage", + logical_id: "triage", + artifact_dir: "artifacts/triage", + outputs: [boundOutput("triaged-findings.json", "ultrafuzz/triaged-findings@1", true)] + }; + const severityNode: PlannedGraphNode = { + ...plannedNode([]), + id: "severity-classification", + logical_id: "severity-classification", + depends_on: [triageNode.id], + artifact_dir: "artifacts/severity-classification", + outputs: [boundOutput("severity-classified-findings.json", "ultrafuzz/severity-classified-findings@1", true)] + }; + const nodes = [triageNode, severityNode]; + writePlannedGraph(layout, nodes); + const triageTask = sealedTaskForNode(layout, triageNode); + const severityTask = sealedTaskForNode(layout, severityNode, [triageTask]); + const tasks = [triageTask, severityTask]; + // The schema requires the severity fields only for true positives, and the + // prompt forbids guessing them for records it keeps but does not promote. + const falsePositive = currentFinding("finding-unreachable", { + status: "false-positive", + triage_classification: "false-positive", + notes: [ + "triage_reason=the state is unreachable through the public path", + "demotion_reason=no public entrypoint reaches the failing state" + ] + }); + writeDeclaredArtifactNode(layout, triageTask.attemptId, triageNode.outputs, { + "triaged-findings.json": JSON.stringify([falsePositive]) + }); + writeDeclaredArtifactNode(layout, severityTask.attemptId, severityNode.outputs, { + "severity-classified-findings.json": JSON.stringify([falsePositive]) + }); + writeSealedFixtureTaskAuthority(layout, nodes, tasks); + + const result = verifyRuntimeRequiredArtifactsForAttempt( + layout, + severityNode, + severityTask.attemptId, + { task: severityTask, tasks }, + authenticatedSnapshotsForNode(layout, severityNode, severityTask.attemptId) + ); + + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); +}); + test("triage gates preserve every deduped finding and upstream note", () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-triage-preservation" }); const dedupeNode: PlannedGraphNode = { @@ -3625,6 +3705,434 @@ for (const markerAuthority of ["dangling leaf", "symlinked root"] as const) { }); } +const AGGREGATION_FIXTURE_GROUPS = { + strategies: { defaults: { failure_policy: "continue" } }, + goals: { defaults: { failure_policy: "continue" } }, + review: {} +}; +const AGGREGATION_DYNAMIC_STORAGE_ID = "dynamic-goal-template-0123456789abcdef0123456789abcdef"; + +interface AggregationFixtureSource { + attemptId: string; + logicalNodeId: string; + manifestPath: string; + manifestSha256: string; + testRelativePath: string; + testPath: string; + testBytes: Buffer; +} + +function aggregationFixtureNode( + id: string, + options: { group?: string; dependsOn?: string[]; outputs?: PlannedGraphNode["outputs"] } = {} +): PlannedGraphNode { + return { + ...plannedNode([]), + id, + logical_id: id, + display_name: id, + ...(options.group === undefined ? {} : { group: options.group }), + depends_on: options.dependsOn ?? [], + artifact_dir: `artifacts/${id}`, + prompt_id: id, + prompt_path: `strategies/${id}.md`, + outputs: options.outputs ?? [boundOutput("generated-tests.json", "ultrafuzz/generated-tests@3", true)] + }; +} + +/** A goal template and one materialized goal, with the identity split `materializeDynamicRuntime` writes. */ +function aggregationDynamicGoalNodes(sourceNodeId: string): { + template: PlannedGraphNode; + generated: PlannedGraphNode; +} { + const template: PlannedGraphNode = { + ...aggregationFixtureNode("goal-template", { group: "goals", dependsOn: [sourceNodeId] }), + dynamic: { + from: { node: sourceNodeId, path: "goal-plan.md" }, + key: "id", + node_id: "dynamic:threat:{{ item.id }}", + status: "pending" + } + }; + const generated: PlannedGraphNode = { + ...template, + id: "dynamic:threat:t1", + display_name: "goal-template: t1", + artifact_dir: `artifacts/${AGGREGATION_DYNAMIC_STORAGE_ID}`, + artifact_dirs: [`artifacts/${AGGREGATION_DYNAMIC_STORAGE_ID}`], + model_fanout: [ + { + attempt_id: AGGREGATION_DYNAMIC_STORAGE_ID, + model_profile_id: "default", + agent_ref: "CodexAgent", + model_name: "gpt-test", + reasoning_effort: "high", + model_index: 0, + loop_index: 0, + attempt_index: 0 + } + ], + workflow: { + node_id: `node:${AGGREGATION_DYNAMIC_STORAGE_ID}`, + task_node_ids: [`node:${AGGREGATION_DYNAMIC_STORAGE_ID}`] + }, + dynamic: undefined, + dynamic_generated: { + group_node_id: template.id, + source_node_id: sourceNodeId, + source_attempt_id: sourceNodeId, + expansion_key: "t1", + item_sha256: "a".repeat(64), + storage_id: AGGREGATION_DYNAMIC_STORAGE_ID, + manifest_path: `dynamic-expansions/${template.id}.json` + } + }; + return { template, generated }; +} + +/** Publish and controller-finalize a generated-tests producer attempt holding one test file. */ +function finalizeGeneratedTestsProducer( + layout: ReturnType, + node: PlannedGraphNode, + attemptId = node.id +): AggregationFixtureSource { + const testRelativePath = `generated-tests/${node.logical_id}.t.sol`; + const testBytes = Buffer.from(`contract GeneratedBy${attemptId.length} {}\n`, "utf8"); + registerArtifactNode(layout, attemptId, node.outputs); + const testPath = writeArtifactFile(layout, attemptId, testRelativePath, testBytes); + const manifestPath = writeArtifactFile( + layout, + attemptId, + "generated-tests.json", + JSON.stringify({ + schema_version: "ultrafuzz.generated-tests.v3", + run_id: layout.runId, + node_id: node.logical_id, + framework: "foundry", + generated_tests: [ + { + path: testRelativePath, + size_bytes: testBytes.byteLength, + sha256: createHash("sha256").update(testBytes).digest("hex") + } + ], + support_files: [] + }) + ); + finalizeArtifactNode(layout, attemptId, node.outputs, { concreteNodeId: node.id, logicalNodeId: node.logical_id }, [ + testRelativePath + ]); + return { + attemptId, + logicalNodeId: node.logical_id, + manifestPath, + manifestSha256: createHash("sha256").update(fs.readFileSync(manifestPath)).digest("hex"), + testRelativePath, + testPath, + testBytes + }; +} + +/** The aggregation manifest an agent writes after copying every listed source bundle into its workspace. */ +function writeAggregationManifestCopying( + layout: ReturnType, + aggregationNode: PlannedGraphNode, + sources: readonly AggregationFixtureSource[] +): void { + const workspace = path.join(layout.workspacesDir, aggregationNode.id); + fs.mkdirSync(workspace, { recursive: true }); + const bundleIdentity = (source: AggregationFixtureSource) => ({ + strategy: source.logicalNodeId, + node_id: source.logicalNodeId, + source_attempt_id: source.attemptId, + attempt_index: 0, + source_manifest_path: source.manifestPath, + source_manifest_relative_path: "generated-tests.json", + source_manifest_sha256: source.manifestSha256 + }); + const files = sources.map((source) => { + const destinationRelativePath = `test/foundry/${source.attemptId}/${path.basename(source.testPath)}`; + const destinationPath = path.join(workspace, ...destinationRelativePath.split("/")); + fs.mkdirSync(path.dirname(destinationPath), { recursive: true }); + fs.writeFileSync(destinationPath, source.testBytes); + return { + ...bundleIdentity(source), + source_artifact_path: source.testPath, + source_relative_path: source.testRelativePath, + destination_path: destinationPath, + destination_relative_path: destinationRelativePath, + size_bytes: source.testBytes.byteLength, + sha256: createHash("sha256").update(source.testBytes).digest("hex") + }; + }); + writeArtifactFile( + layout, + aggregationNode.id, + "aggregation.json", + JSON.stringify({ + schema_version: "ultrafuzz.aggregation-manifest.v1", + source_generated_tests: sources.length, + copied_generated_tests: sources.length, + source_support_files: 0, + copied_support_files: 0, + source_bundles: sources.map((source) => ({ + ...bundleIdentity(source), + source_run_id: layout.runId, + framework: "foundry", + generated_test_count: 1, + support_file_count: 0, + disposition: "copied" + })), + files, + support_files: [], + skipped_files: [] + }) + ); +} + +/** Run the host gate the way finalization does: on the persisted graph node with its sealed task set. */ +function verifyAggregationAttempt( + layout: ReturnType, + aggregationTask: SmithersTaskManifestTask, + tasks: SmithersTaskManifestTask[], + admittedDependencyAttemptIds: string[] +): ReturnType { + const graph = JSON.parse(fs.readFileSync(layout.graphPath, "utf8")) as PlannedGraph; + const node = graph.nodes.find((candidate) => candidate.id === aggregationTask.concreteNodeId); + assert.ok(node); + return verifyRuntimeRequiredArtifactsForAttempt( + layout, + node, + aggregationTask.attemptId, + { task: aggregationTask, tasks, admittedDependencyAttemptIds }, + authenticatedSnapshotsForNode(layout, node, aggregationTask.attemptId) + ); +} + +test("host aggregation intake skips a failed optional generated-tests producer the verifier did not admit", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-aggregation-failed-optional" }); + const verified = aggregationFixtureNode("strategy-a", { group: "strategies" }); + const failed = aggregationFixtureNode("strategy-b", { group: "strategies" }); + const aggregationNode = aggregationFixtureNode("aggregate-test-files", { + group: "review", + dependsOn: [verified.id, failed.id], + outputs: [boundOutput("aggregation.json", "ultrafuzz/aggregation-manifest@1", true)] + }); + const nodes = [verified, failed, aggregationNode]; + writePlannedGraph(layout, nodes, AGGREGATION_FIXTURE_GROUPS); + const verifiedTask = sealedTaskForNode(layout, verified); + const failedTask = sealedTaskForNode(layout, failed); + const aggregationTask: SmithersTaskManifestTask = { + ...sealedTaskForNode(layout, aggregationNode, [verifiedTask, failedTask]), + optionalDependencyArtifactDirs: [verifiedTask.artifactDir, failedTask.artifactDir] + }; + const tasks = [verifiedTask, failedTask, aggregationTask]; + const source = finalizeGeneratedTestsProducer(layout, verified); + // The continue-policy strategy failed before its verifier wrote a marker. + registerArtifactNode(layout, failedTask.attemptId, failed.outputs); + updateNodeState(layout, failedTask.attemptId, { status: "failed" }); + writeSealedFixtureTaskAuthority(layout, nodes, tasks); + + writeAggregationManifestCopying(layout, aggregationNode, [source]); + const result = verifyAggregationAttempt(layout, aggregationTask, tasks, [verifiedTask.attemptId]); + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); + + // The admitted producer is still part of the authority: omitting it fails. + writeAggregationManifestCopying(layout, aggregationNode, []); + const omitted = verifyAggregationAttempt(layout, aggregationTask, tasks, [verifiedTask.attemptId]); + assert.equal(omitted.ok, false); + assert.ok( + omitted.diagnostics.some((diagnostic) => diagnostic.message.includes("omits authenticated source bundle")), + JSON.stringify(omitted.diagnostics) + ); + + // An admitted producer is read only once the controller finalized it. + writeAggregationManifestCopying(layout, aggregationNode, [source]); + fs.unlinkSync(path.join(layout.root, ".ultrafuzz-verification", `${verifiedTask.attemptId}.json`)); + const unfinalized = verifyAggregationAttempt(layout, aggregationTask, tasks, [verifiedTask.attemptId]); + assert.equal(unfinalized.ok, false); + assert.ok( + unfinalized.diagnostics.some( + (diagnostic) => + diagnostic.code === "REQUIRED_ARTIFACT_INVALID" && + diagnostic.message.includes(`authority is invalid for ${verifiedTask.attemptId}`) + ), + JSON.stringify(unfinalized.diagnostics) + ); +}); + +test("host aggregation intake keys a directly consumed dynamic producer by its storage attempt ID", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-aggregation-direct-dynamic" }); + const source = aggregationFixtureNode("goal-plan", { + outputs: [boundOutput("goal-plan.md", "ultrafuzz/nonempty-markdown@1", true)] + }); + const { template, generated } = aggregationDynamicGoalNodes(source.id); + const aggregationNode = aggregationFixtureNode("aggregate-test-files", { + group: "review", + dependsOn: [generated.id], + outputs: [boundOutput("aggregation.json", "ultrafuzz/aggregation-manifest@1", true)] + }); + const nodes = [source, template, generated, aggregationNode]; + writePlannedGraph(layout, nodes, AGGREGATION_FIXTURE_GROUPS); + const sourceTask = sealedTaskForNode(layout, source); + const generatedTask = sealedTaskForNode(layout, generated, [sourceTask], AGGREGATION_DYNAMIC_STORAGE_ID); + const aggregationTask: SmithersTaskManifestTask = { + ...sealedTaskForNode(layout, aggregationNode, [generatedTask]), + optionalDependencyArtifactDirs: [generatedTask.artifactDir] + }; + const tasks = [sourceTask, generatedTask, aggregationTask]; + writeDeclaredArtifactNode(layout, sourceTask.attemptId, source.outputs, { "goal-plan.md": "# Goal plan\n" }); + const dynamicSource = finalizeGeneratedTestsProducer(layout, generated, AGGREGATION_DYNAMIC_STORAGE_ID); + writeSealedFixtureTaskAuthority(layout, nodes, tasks); + + writeAggregationManifestCopying(layout, aggregationNode, [dynamicSource]); + const result = verifyAggregationAttempt(layout, aggregationTask, tasks, [ + sourceTask.attemptId, + AGGREGATION_DYNAMIC_STORAGE_ID + ]); + + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); +}); + +test("host aggregation intake attributes model-fanout bundles to the loop attempt index the verifier uses", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-aggregation-model-fanout" }); + const fanout: PlannedGraphNode = { + ...aggregationFixtureNode("strategy-a", { group: "strategies" }), + model_fanout: [0, 1].map((modelIndex) => ({ + model_profile_id: `model-${modelIndex}`, + agent_ref: "CodexAgent", + model_name: "gpt-test", + reasoning_effort: "high", + model_index: modelIndex, + loop_index: 0, + attempt_index: modelIndex + })) + }; + const aggregationNode = aggregationFixtureNode("aggregate-test-files", { + group: "review", + dependsOn: [fanout.id], + outputs: [boundOutput("aggregation.json", "ultrafuzz/aggregation-manifest@1", true)] + }); + const nodes = [fanout, aggregationNode]; + writePlannedGraph(layout, nodes, AGGREGATION_FIXTURE_GROUPS); + const fanoutTasks = [0, 1].map((modelIndex) => + smithersTaskForNode({ + layout, + node: fanout, + attemptId: `${fanout.id}__model_${modelIndex}__attempt_${modelIndex}`, + modelIndex + }) + ); + const aggregationTask: SmithersTaskManifestTask = { + ...sealedTaskForNode(layout, aggregationNode, fanoutTasks), + optionalDependencyArtifactDirs: fanoutTasks.map((task) => task.artifactDir) + }; + const tasks = [...fanoutTasks, aggregationTask]; + const sources = fanoutTasks.map((task) => { + const source = finalizeGeneratedTestsProducer(layout, fanout, task.attemptId); + // The controller records the model attempt index (here 0 and 1) in each + // producer manifest; the verifier attributes both bundles to loop attempt 0. + const manifestPath = path.join(task.artifactDir, "artifact-manifest.json"); + const manifest = JSON.parse(fs.readFileSync(manifestPath, "utf8")) as { provenance: { attempt_index?: number } }; + manifest.provenance.attempt_index = task.metadata.model.attemptIndex; + fs.writeFileSync(manifestPath, JSON.stringify(manifest)); + const state = readRunState(layout); + const provenance = state.nodes[task.attemptId]?.provenance; + assert.ok(provenance !== undefined && "output_contracts" in provenance && provenance.output_contracts); + provenance.output_contracts.artifact_manifest_sha256 = createHash("sha256") + .update(fs.readFileSync(manifestPath)) + .digest("hex"); + writeRunState(layout, state); + return source; + }); + writeSealedFixtureTaskAuthority(layout, nodes, tasks); + + writeAggregationManifestCopying(layout, aggregationNode, sources); + const result = verifyAggregationAttempt( + layout, + aggregationTask, + tasks, + fanoutTasks.map((task) => task.attemptId) + ); + + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); +}); + +test("host aggregation intake follows the sealed closure below a dynamic group's direct dependent", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-aggregation-transitive-dynamic" }); + const source = aggregationFixtureNode("goal-plan", { + outputs: [boundOutput("goal-plan.md", "ultrafuzz/nonempty-markdown@1", true)] + }); + const { template, generated } = aggregationDynamicGoalNodes(source.id); + const join = aggregationFixtureNode("goal-join", { + group: "review", + dependsOn: [generated.id], + outputs: [boundOutput("goal-join.md", "ultrafuzz/nonempty-markdown@1", true)] + }); + const strategy = aggregationFixtureNode("strategy-a", { group: "strategies" }); + const unrelated = aggregationFixtureNode("strategy-z", { group: "strategies" }); + const aggregationNode = aggregationFixtureNode("aggregate-test-files", { + group: "review", + dependsOn: [join.id, strategy.id], + outputs: [boundOutput("aggregation.json", "ultrafuzz/aggregation-manifest@1", true)] + }); + const nodes = [source, template, generated, join, strategy, unrelated, aggregationNode]; + writePlannedGraph(layout, nodes, AGGREGATION_FIXTURE_GROUPS); + const sourceTask = sealedTaskForNode(layout, source); + const generatedTask = sealedTaskForNode(layout, generated, [sourceTask], AGGREGATION_DYNAMIC_STORAGE_ID); + const joinTask: SmithersTaskManifestTask = { + ...sealedTaskForNode(layout, join, [generatedTask]), + optionalDependencyArtifactDirs: [generatedTask.artifactDir] + }; + const strategyTask = sealedTaskForNode(layout, strategy); + const unrelatedTask = sealedTaskForNode(layout, unrelated); + // Dynamic lowering extends only a group's direct dependents, so the sealed + // closure of this grandchild keeps its compile-time ancestors and never + // names the generated attempt that the lowered planned graph reaches. + const aggregationTaskWithClosure = (closure: readonly SmithersTaskManifestTask[]): SmithersTaskManifestTask => ({ + ...smithersTaskForNode({ + layout, + node: aggregationNode, + attemptId: aggregationNode.id, + dependencies: [joinTask.attemptId, strategyTask.attemptId], + dependencyArtifactDirs: closure.map((task) => task.artifactDir) + }), + optionalDependencyArtifactDirs: closure + .filter((task) => task.metadata.node.group === "strategies") + .map((task) => task.artifactDir) + }); + const aggregationTask = aggregationTaskWithClosure([sourceTask, joinTask, strategyTask]); + const tasks = [sourceTask, generatedTask, joinTask, strategyTask, unrelatedTask, aggregationTask]; + writeDeclaredArtifactNode(layout, sourceTask.attemptId, source.outputs, { "goal-plan.md": "# Goal plan\n" }); + finalizeGeneratedTestsProducer(layout, generated, AGGREGATION_DYNAMIC_STORAGE_ID); + const strategySource = finalizeGeneratedTestsProducer(layout, strategy); + writeSealedFixtureTaskAuthority(layout, nodes, tasks); + + writeAggregationManifestCopying(layout, aggregationNode, [strategySource]); + const admitted = [sourceTask.attemptId, joinTask.attemptId, strategyTask.attemptId]; + const result = verifyAggregationAttempt(layout, aggregationTask, tasks, admitted); + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); + + // Only generated attempts may be absent; the closure still may not reach + // outside the planned ancestors. + const widenedTask = aggregationTaskWithClosure([sourceTask, joinTask, strategyTask, unrelatedTask]); + const widened = verifyAggregationAttempt( + layout, + widenedTask, + tasks.map((task) => (task === aggregationTask ? widenedTask : task)), + admitted + ); + assert.equal(widened.ok, false); + assert.ok( + widened.diagnostics.some((diagnostic) => + diagnostic.message.includes( + `artifact ancestor closure does not match the exact planned set; unexpected: ${JSON.stringify(unrelatedTask.artifactDir)}` + ) + ), + JSON.stringify(widened.diagnostics) + ); +}); + test("review lifecycle and strategy gates authenticate every dedupe, triage, and severity transition", () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-review-lifecycle" }); const artifactPaths = { @@ -3995,7 +4503,12 @@ test("sealed planned graph ignores unrelated producers but rejects a missing pla writePlannedGraph(outsideLayout, [outsideCatalogNode, reportNode]); writeArtifact(outsideLayout, reportNode.id, "report.md", "# Report\n"); writeArtifact(outsideLayout, reportNode.id, "report.json", JSON.stringify(currentReport(outsideLayout.runId))); - const outside = verifyRuntimeRequiredArtifactsForAttempt(outsideLayout, reportNode, reportNode.id); + const outside = verifyRuntimeRequiredArtifactsForAttempt( + outsideLayout, + reportNode, + reportNode.id, + sealedFixtureAuthority(outsideLayout, reportNode.id) + ); assert.equal(outside.ok, true, JSON.stringify(outside.diagnostics)); const missingLayout = createRunLayout({ projectRoot: tempProject(), runId: "run-missing-property-producer" }); @@ -4674,16 +5187,32 @@ test("project discovery gate requires ledger evidence to survive in the markdown ); }); -test("artifact validation rejects a persisted schema binding that differs from the current registry", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-schema-binding-mismatch" }); +test("artifact validation binds schema content and treats the validator build as provenance", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-schema-binding-provenance" }); const artifactDir = getNodeArtifactDir(layout, "strategy-a", { create: true }); const node = plannedNode(["findings.json"]); - node.outputs[0]!.schema_sha256 = "0".repeat(64); + const [findings] = node.outputs; + assert.ok(findings); + // #921: an output planned by another validator build of the same schemas is still gated on its + // content. A validator rebuild after launch must neither pass an invalid artifact nor fail a valid one. + findings.validator_build = `ultrafuzz-json-validator.v1:${"9".repeat(64)}`; fs.writeFileSync(path.join(artifactDir, "findings.json"), "[]\n", "utf8"); + const rebuilt = verifyRequiredArtifactsForAttempt(layout, node, node.id); + assert.equal(rebuilt.ok, true, JSON.stringify(rebuilt.diagnostics)); + fs.writeFileSync(path.join(artifactDir, "findings.json"), "{}\n", "utf8"); + const invalid = verifyRequiredArtifactsForAttempt(layout, node, node.id); + assert.equal(invalid.ok, false); + assert.ok(invalid.diagnostics.some((diagnostic) => diagnostic.code === "JSON_SCHEMA_VIOLATION")); - const result = verifyRequiredArtifactsForAttempt(layout, node, node.id); - assert.equal(result.ok, false); - assert.ok(result.diagnostics.some((diagnostic) => diagnostic.code === "ARTIFACT_SCHEMA_BINDING_MISMATCH")); + // A planned schema digest that the planned bundle does not contain cannot be validated at all. + fs.writeFileSync(path.join(artifactDir, "findings.json"), "[]\n", "utf8"); + findings.schema_sha256 = "0".repeat(64); + const unknownSchema = verifyRequiredArtifactsForAttempt(layout, node, node.id); + assert.equal(unknownSchema.ok, false); + assert.ok( + unknownSchema.diagnostics.some((diagnostic) => diagnostic.code === "ARTIFACT_VALIDATOR_IDENTITY_MISMATCH"), + JSON.stringify(unknownSchema.diagnostics) + ); }); test("project discovery gate accepts a repository-root scan probe", () => { @@ -4761,6 +5290,29 @@ test("project discovery gate rejects an empty ledger that does not justify the a const justified = verifyRequiredArtifactsForAttempt(layout, node, node.id); assert.equal(justified.ok, true, JSON.stringify(justified.diagnostics)); + // The ledger schema is the only check for the remaining incomplete shapes. + const justifiedLedger = JSON.parse(ledger("The target states no invariant.")) as Record; + for (const [label, document] of [ + ["missing inventory_rows", { ...justifiedLedger, inventory_rows: undefined }], + ["missing scan_probes", { ...justifiedLedger, scan_probes: undefined }], + ["empty ledger without scan probes", { ...justifiedLedger, scan_probes: [] }], + [ + "empty ledger with inventory rows", + { + ...justifiedLedger, + inventory_rows: [{ id: "inventory-1", description: "Solvency", ledger_ids: ["evidence-1"] }] + } + ] + ] as const) { + writeArtifact(layout, "project-discovery", "setup/invariant-evidence-ledger.json", JSON.stringify(document)); + const incomplete = verifyRequiredArtifactsForAttempt(layout, node, node.id); + assert.equal(incomplete.ok, false, label); + assert.ok( + incomplete.diagnostics.some((diagnostic) => diagnostic.code === "JSON_SCHEMA_VIOLATION"), + `${label}: ${JSON.stringify(incomplete.diagnostics)}` + ); + } + // A justification on a ledger that DOES carry entries is contradictory, so the schema refuses it // rather than letting both readings of the artifact coexist. writeArtifact( @@ -6329,12 +6881,16 @@ for (const [label, undeclaredPath] of [ ["conventional", "properties/recon.json"], ["arbitrary sibling", "properties/other.json"] ] as const) { - test(`property fan-in ignores ${label} lens JSON without a state declaration`, () => { + test(`property fan-in ignores ${label} lens JSON without a sealed lens declaration`, () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: `run-undeclared-lens-${label.replaceAll(" ", "-")}` }); const node = writeMinimalPropertyFaninFixture(layout); + // The catalog source's planned (and therefore sealed) outputs declare no property lens. + registerArtifactNode(layout, "property-specification-recon", [ + boundOutput("notes.md", "ultrafuzz/nonempty-markdown@1", true) + ]); const lensPath = writeArtifact( layout, "property-specification-recon", @@ -6762,8 +7318,8 @@ test("property fan-in selects a declared lens contract without relying on the pr assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); }); -test("property fan-in cannot hide a planned lens by omitting its state declaration and catalog rows", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-planned-lens-state-omission" }); +test("property fan-in cannot hide a planned lens by omitting its catalog rows", () => { + const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-planned-lens-catalog-omission" }); const node = writeMinimalPropertyFaninFixture(layout, { sourceNodeId: "project-discovery", sourcePropertyId: "evidence-1", @@ -6785,15 +7341,13 @@ test("property fan-in cannot hide a planned lens by omitting its state declarati ] }) ); - const state = readRunState(layout); - delete state.nodes["property-specification-recon"]!.outputs; - writeRunState(layout, state); const result = verifyRequiredArtifactsForAttempt(layout, node, node.id); assert.equal(result.ok, false, JSON.stringify(result.diagnostics)); - assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_LENS_DECLARATION_MISSING"), + assert.deepEqual( + gateIssuePaths(result, "property-source-join"), + ["$.properties"], JSON.stringify(result.diagnostics) ); }); @@ -6949,16 +7503,10 @@ test("property fan-in rejects ambiguous property-lens declarations", () => { ); }); -test("property fan-in rejects a stale property-lens schema binding", () => { +test("property fan-in rejects a sealed lens declaration whose schema binding differs from the plan", () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-stale-declared-lens" }); const node = writeMinimalPropertyFaninFixture(layout); - registerArtifactNode(layout, "property-specification-recon", [ - { - ...boundOutput("custom/recon.json", "ultrafuzz/property-lens@2"), - schema_sha256: "0".repeat(64) - } - ]); - writeArtifact( + writeDeclaredPropertyLens( layout, "property-specification-recon", "custom/recon.json", @@ -6974,12 +7522,35 @@ test("property fan-in rejects a stale property-lens schema binding", () => { ] }) ); + const current = verifyRequiredArtifactsForAttempt(layout, node, node.id); + assert.equal(current.ok, true, JSON.stringify(current.diagnostics)); - const result = verifyRequiredArtifactsForAttempt(layout, node, node.id); + const sealed = JSON.parse( + fs.readFileSync(path.join(layout.root, "smithers", "tasks.json"), "utf8") + ) as SmithersTaskManifestDocument; + const staleTasks = sealed.tasks.map((task) => + task.attemptId !== "property-specification-recon" + ? task + : { + ...task, + metadata: { + ...task.metadata, + artifacts: { + ...task.metadata.artifacts, + outputs: task.metadata.artifacts.outputs.map((output) => ({ ...output, schemaSha256: "0".repeat(64) })) + } + } + } + ); + const result = verifyRequiredArtifactsForAttempt(layout, node, node.id, fixtureAttemptAuthority(staleTasks, node.id)); assert.equal(result.ok, false, JSON.stringify(result.diagnostics)); assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_LENS_SCHEMA_BINDING_INVALID"), + result.diagnostics.some((diagnostic) => + diagnostic.message.includes( + "sealed Smithers attempt does not match its planned node and outputs: property-specification-recon" + ) + ), JSON.stringify(result.diagnostics) ); }); @@ -7951,12 +8522,11 @@ test("property consumers reject a finalized catalog whose typed Markdown compani ); }); -test("property implementation accepts an intentional producer-free empty catalog through the full host gate", () => { - const layout = createRunLayout({ - projectRoot: tempProject(), - runId: "run-producer-free-implementation", - resolvedConfigToml: '[invariants]\nproperty_priority_threshold = "high"\n' - }); +function producerFreeImplementationGate( + runId: string, + resolvedConfigToml: string +): ReturnType { + const layout = createRunLayout({ projectRoot: tempProject(), runId, resolvedConfigToml }); const output = boundOutput("handoffs/implementation-v3.json", "ultrafuzz/implemented-properties@3", true); const node: PlannedGraphNode = { ...plannedNode([]), @@ -7975,12 +8545,32 @@ test("property implementation accepts an intentional producer-free empty catalog }) }); writePlannedGraph(layout, [node]); + return verifyRuntimeRequiredArtifactsForAttempt(layout, node, node.id, sealedFixtureAuthority(layout, node.id)); +} - const result = verifyRuntimeRequiredArtifactsForAttempt(layout, node, node.id); - +test("property implementation accepts an intentional producer-free empty catalog through the full host gate", () => { + const result = producerFreeImplementationGate( + "run-producer-free-implementation", + '[invariants]\nproperty_priority_threshold = "high"\n' + ); assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); }); +// The gate reads config.resolved.toml with the project config parser, so any +// valid TOML spelling of the setting counts, not only one line shape. +test("property implementation reads the resolved priority threshold as TOML", () => { + for (const [label, resolvedConfigToml] of [ + ["trailing comment", '[invariants]\nproperty_priority_threshold = "high" # operator note\n'], + ["inline table", 'invariants = { property_priority_threshold = "high" }\n'] + ] as const) { + const result = producerFreeImplementationGate( + `run-priority-toml-${label.replaceAll(" ", "-")}`, + resolvedConfigToml + ); + assert.equal(result.ok, true, `${label}: ${JSON.stringify(result.diagnostics)}`); + } +}); + test("property implementation gate enforces declared selection coverage and actionable blockers", () => { const layout = createRunLayout({ projectRoot: tempProject(), @@ -8395,6 +8985,19 @@ test("property implementation gate rejects an unknown finding property reference ); }); +/** A registry gate's failures in a host gate result, by in-document path. */ +function gateIssuePaths(result: ReturnType, gate: string): string[] { + return result.diagnostics + .filter((diagnostic) => diagnostic.details?.gate === gate) + .map((diagnostic) => diagnostic.path?.slice(diagnostic.path.indexOf("#") + 1) ?? ""); +} + +function campaignJoinIssues(result: ReturnType): string[] { + return result.diagnostics + .filter((diagnostic) => diagnostic.details?.gate === "property-campaign-context-joins") + .map((diagnostic) => `${diagnostic.path?.slice(diagnostic.path.indexOf("#") + 1)}: ${diagnostic.message}`); +} + test("campaign gate accepts non-property findings and validates property-derived failures", () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-campaign" }); writeArtifact( @@ -8469,7 +9072,10 @@ test("campaign gate accepts non-property findings and validates property-derived ); const dropped = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(dropped.ok, false); - assert.ok(dropped.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_FINDING_REFERENCE_MISMATCH")); + assert.ok( + gateIssuePaths(dropped, "property-campaign-context-joins").includes("$.findings_ref#0.property_ids"), + JSON.stringify(dropped.diagnostics) + ); writeArtifact( layout, @@ -8491,7 +9097,10 @@ test("campaign gate accepts non-property findings and validates property-derived ); const unknown = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(unknown.ok, false); - assert.ok(unknown.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_REFERENCE_UNKNOWN")); + assert.ok( + gateIssuePaths(unknown, "property-campaign-context-joins").includes("$.failures[0].property_ids[0]"), + JSON.stringify(unknown.diagnostics) + ); writeArtifact( layout, @@ -8657,7 +9266,10 @@ test("campaign gate still applies to project-owned split recon campaign nodes", const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); - assert.ok(result.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_REFERENCE_UNKNOWN")); + assert.ok( + gateIssuePaths(result, "property-campaign-context-joins").includes("$.failures[0].property_ids[0]"), + JSON.stringify(result.diagnostics) + ); }); test("campaign gates select custom declared paths and ignore undeclared conventional files", () => { @@ -8778,14 +9390,42 @@ test("producer-free final reports require the exact not-planned implementation c }); writePlannedGraph(layout, [node]); - const valid = verifyRuntimeRequiredArtifactsForAttempt(layout, node, node.id); + const valid = verifyRuntimeRequiredArtifactsForAttempt( + layout, + node, + node.id, + sealedFixtureAuthority(layout, node.id) + ); assert.equal(valid.ok, true, JSON.stringify(valid.diagnostics)); + for (const [prose, code] of [ + ["Recon reached 85% line coverage on Vault.sol.", "UNSCOPED_COVERAGE_PERCENTAGE"], + ["Coverage of withdraw() was 3/4 branches in the replay.", "UNSCOPED_COVERAGE_FRACTION"] + ] as const) { + writeDeclaredArtifactNode(layout, node.id, outputs, { + "deliverables/report.md": `# Ultrafuzz report\n\n${prose}\n`, + "deliverables/report.json": JSON.stringify(currentReport(layout.runId)) + }); + const unscopedProse = verifyRuntimeRequiredArtifactsForAttempt( + layout, + node, + node.id, + sealedFixtureAuthority(layout, node.id) + ); + assert.equal(unscopedProse.ok, true, `${prose}: ${JSON.stringify(unscopedProse.diagnostics)}`); + assertAdvisoryCoverageScore(unscopedProse, code, prose); + } + writeDeclaredArtifactNode(layout, node.id, outputs, { "deliverables/report.md": "# Ultrafuzz report\n\nrecon-selected-declaration-completeness: `1/1`\n", "deliverables/report.json": JSON.stringify(currentReport(layout.runId)) }); - const inventedCoverage = verifyRuntimeRequiredArtifactsForAttempt(layout, node, node.id); + const inventedCoverage = verifyRuntimeRequiredArtifactsForAttempt( + layout, + node, + node.id, + sealedFixtureAuthority(layout, node.id) + ); assert.equal(inventedCoverage.ok, false, JSON.stringify(inventedCoverage.diagnostics)); assert.ok( inventedCoverage.diagnostics.some((diagnostic) => diagnostic.code === "REPORT_COVERAGE_EVIDENCE_UNPLANNED"), @@ -8800,7 +9440,12 @@ test("producer-free final reports require the exact not-planned implementation c }) ) }); - const mismatched = verifyRuntimeRequiredArtifactsForAttempt(layout, node, node.id); + const mismatched = verifyRuntimeRequiredArtifactsForAttempt( + layout, + node, + node.id, + sealedFixtureAuthority(layout, node.id) + ); assert.equal(mismatched.ok, false, JSON.stringify(mismatched.diagnostics)); assert.ok( mismatched.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_REPORT_IMPLEMENTATION_COVERAGE_MISMATCH"), @@ -8885,6 +9530,18 @@ test("final reports disclose planned but omitted implementation coverage without fs.existsSync(path.join(layout.artifactsDir, optionalTask.attemptId, "implemented-properties.json")), false ); + + const prose = "Recon reached 85% line coverage on Vault.sol."; + writeArtifactFile(layout, reportTask.attemptId, "report.md", `# Ultrafuzz report\n\n${prose}\n`); + const unadmittedCoverageProse = verifyRuntimeRequiredArtifactsForAttempt( + layout, + reportNode, + reportTask.attemptId, + { task: reportTask, tasks, admittedDependencyAttemptIds: [] }, + authenticatedSnapshotsForNode(layout, reportNode, reportTask.attemptId) + ); + assert.equal(unadmittedCoverageProse.ok, true, JSON.stringify(unadmittedCoverageProse.diagnostics)); + assertAdvisoryCoverageScore(unadmittedCoverageProse, "UNSCOPED_COVERAGE_PERCENTAGE", prose); }); test("final report gate joins the default recon-only campaign backend", () => { @@ -10112,7 +10769,7 @@ interface CampaignTimeoutResultFixture extends Record { schema_version: "ultrafuzz.property-campaign.v3"; fuzzer_backend: "recon"; configured_timeout_seconds: number; - sequence_length: number; + sequence_length?: number; exact_command: string; start_timestamp: string; end_timestamp: string; @@ -10268,8 +10925,6 @@ function runCampaignTimeoutGate( topologyTimeoutSeconds?: number | null; modelTimeoutSeconds?: number | null; logicalNodeId?: string; - outputCounts?: Partial>; - nonRecordDocument?: "result" | "summary"; declaredPaths?: { plan: string; result: string; @@ -10286,15 +10941,6 @@ function runCampaignTimeoutGate( findings: "findings.json", summary: "campaign-summary.json" }; - const outputCounts = { - plan: 1, - result: 1, - findings: 1, - summary: 1, - ...options.outputCounts - }; - const rolePath = (role: keyof typeof outputCounts, index: number): string => - index === 0 ? declaredPaths[role] : `${declaredPaths[role]}.duplicate-${index}`; const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-campaign-timeout", @@ -10321,7 +10967,7 @@ function runCampaignTimeoutGate( layout, campaignId, declaredPaths.result, - JSON.stringify(options.nonRecordDocument === "result" ? [] : fixture.backend), + JSON.stringify(fixture.backend), "ultrafuzz/property-campaign@3" ); writeArtifact(layout, campaignId, declaredPaths.findings, "[]", "ultrafuzz/findings@2"); @@ -10329,60 +10975,15 @@ function runCampaignTimeoutGate( layout, campaignId, declaredPaths.summary, - JSON.stringify(options.nonRecordDocument === "summary" ? [] : fixture.summary), + JSON.stringify(fixture.summary), "ultrafuzz/campaign-summary@2" ); - for (let index = 1; index < outputCounts.plan; index += 1) { - writeArtifact( - layout, - campaignId, - rolePath("plan", index), - JSON.stringify(fixture.plan), - "ultrafuzz/invariant-campaign-plan@2" - ); - } - for (let index = 1; index < outputCounts.result; index += 1) { - writeArtifact( - layout, - campaignId, - rolePath("result", index), - JSON.stringify(fixture.backend), - "ultrafuzz/property-campaign@3" - ); - } - for (let index = 1; index < outputCounts.findings; index += 1) { - writeArtifact(layout, campaignId, rolePath("findings", index), "[]", "ultrafuzz/findings@2"); - } - for (let index = 1; index < outputCounts.summary; index += 1) { - writeArtifact( - layout, - campaignId, - rolePath("summary", index), - JSON.stringify(fixture.summary), - "ultrafuzz/campaign-summary@2" - ); - } const base = currentCampaignNode([ "campaign-plan.json", "recon-fuzzer-results.json", "findings.json", "campaign-summary.json" ]); - const outputs = [ - ...Array.from({ length: outputCounts.plan }, (_, index) => - boundOutput(rolePath("plan", index), "ultrafuzz/invariant-campaign-plan@2", index === 0) - ), - ...Array.from({ length: outputCounts.result }, (_, index) => - boundOutput(rolePath("result", index), "ultrafuzz/property-campaign@3", false) - ), - ...Array.from({ length: outputCounts.findings }, (_, index) => - boundOutput(rolePath("findings", index), "ultrafuzz/findings@2", false) - ), - ...Array.from({ length: outputCounts.summary }, (_, index) => - boundOutput(rolePath("summary", index), "ultrafuzz/campaign-summary@2", false) - ) - ]; - if (outputs.length > 0 && !outputs.some((output) => output.primary)) outputs[0] = { ...outputs[0]!, primary: true }; const node: PlannedGraphNode = { ...base, id: campaignId, @@ -10403,19 +11004,35 @@ function runCampaignTimeoutGate( } ] }), - outputs + outputs: [ + boundOutput(declaredPaths.plan, "ultrafuzz/invariant-campaign-plan@2", true), + boundOutput(declaredPaths.result, "ultrafuzz/property-campaign@3"), + boundOutput(declaredPaths.findings, "ultrafuzz/findings@2"), + boundOutput(declaredPaths.summary, "ultrafuzz/campaign-summary@2") + ] }; if (topologyTimeoutSeconds === null) delete node.timeout_seconds; return verifyRequiredArtifactsForAttempt(layout, node, campaignId); } +/** + * Registry gate property-campaign-timeout-evidence failures, by in-document + * path. The host runs the same gate the generated verifier runs; there is no + * separate host implementation of these rules. + */ +function timeoutEvidenceIssuePaths(result: ReturnType): string[] { + return gateIssuePaths(result, "property-campaign-timeout-evidence"); +} + +function withCampaignCommand(fixture: CampaignTimeoutFixture, command: string): void { + fixture.backend.exact_command = command; + fixture.plan.backend.exact_shell_escaped_command = command; + fixture.plan.command_plan = [{ phase: "campaign", command }]; +} + test("current campaign timeout gate accepts exact configured Recon timeout evidence", () => { const result = runCampaignTimeoutGate(); assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); - assert.deepEqual( - result.diagnostics.filter((diagnostic) => diagnostic.source === "campaign-timeout-evidence"), - [] - ); }); test("current campaign timeout gate resolves custom artifact paths from sealed contract declarations", () => { @@ -10437,58 +11054,24 @@ test("current campaign timeout gate resolves custom artifact paths from sealed c assert.ok( result.diagnostics.some( (diagnostic) => - diagnostic.code === "CAMPAIGN_TIMEOUT_CONFIG_MISMATCH" && - diagnostic.path?.endsWith("custom/campaign-plan.json#$.configured_fuzzer_timeout_seconds") + diagnostic.details?.gate === "property-campaign-timeout-evidence" && + diagnostic.path?.endsWith("custom/recon-results.json#$.campaign_plan_ref#configured_fuzzer_timeout_seconds") ), JSON.stringify(result.diagnostics) ); }); -test("current campaign timeout gate requires exactly one declaration for every tuple member", () => { - const tupleMembers = [ - ["plan", "ultrafuzz/invariant-campaign-plan@2"], - ["result", "ultrafuzz/property-campaign@3"], - ["summary", "ultrafuzz/campaign-summary@2"], - ["findings", "ultrafuzz/findings@2"] - ] as const; - for (const [role, contract] of tupleMembers) { - for (const count of [0, 2] as const) { - const result = runCampaignTimeoutGate(() => undefined, { - logicalNodeId: "project-owned-recon-campaign", - outputCounts: { [role]: count } - }); - assert.equal(result.ok, false, `${role}:${count}`); - assert.ok( - result.diagnostics.some( - (diagnostic) => - diagnostic.code === "CAMPAIGN_TIMEOUT_OUTPUT_DECLARATION_INVALID" && - diagnostic.message.includes(contract) && - diagnostic.message.endsWith(`found ${count}`) - ), - `${role}:${count}: ${JSON.stringify(result.diagnostics)}` - ); - } - } -}); - -test("current campaign timeout gate explicitly rejects non-object result and summary documents", () => { - for (const role of ["result", "summary"] as const) { - const result = runCampaignTimeoutGate(() => undefined, { nonRecordDocument: role }); - assert.equal(result.ok, false, role); - assert.ok( - result.diagnostics.some( - (diagnostic) => - diagnostic.code === "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID" && diagnostic.message.includes("must be an object") - ), - `${role}: ${JSON.stringify(result.diagnostics)}` - ); - } -}); - test("current campaign timeout gate requires the sealed topology node budget", () => { const result = runCampaignTimeoutGate(() => undefined, { topologyTimeoutSeconds: null }); assert.equal(result.ok, false); - assert.ok(result.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_PLAN_BUDGET_MISSING")); + assert.ok( + result.diagnostics.some( + (diagnostic) => + diagnostic.code === "ARTIFACT_SEMANTIC_GATE_CONTEXT_UNAVAILABLE" && + diagnostic.details?.gate === "property-campaign-timeout-evidence" + ), + JSON.stringify(result.diagnostics) + ); }); test("current campaign timeout gate accepts a sealed model-profile or run-default budget", () => { @@ -10496,412 +11079,224 @@ test("current campaign timeout gate accepts a sealed model-profile or run-defaul topologyTimeoutSeconds: null, modelTimeoutSeconds: 7200 }); - assert.deepEqual( - result.diagnostics.filter((diagnostic) => diagnostic.source === "campaign-timeout-evidence"), - [], - JSON.stringify(result.diagnostics) - ); + assert.deepEqual(timeoutEvidenceIssuePaths(result), [], JSON.stringify(result.diagnostics)); }); -test("current campaign timeout gate rejects reserve subtraction and ambiguous Recon command flags", () => { - const cases: Array<{ - name: string; - code: string; - mutate: (fixture: CampaignTimeoutFixture) => void; - }> = [ - { - name: "plan configured timeout", - code: "CAMPAIGN_TIMEOUT_CONFIG_MISMATCH", - mutate: (fixture) => { - fixture.plan.configured_fuzzer_timeout_seconds = 3300; - } - }, - { - name: "Recon internal timeout", - code: "CAMPAIGN_TIMEOUT_CONFIG_MISMATCH", - mutate: (fixture) => { - fixture.plan.recon_internal_timeout_seconds = 3300; - } - }, - { - name: "host soft timeout", - code: "CAMPAIGN_TIMEOUT_CONFIG_MISMATCH", - mutate: (fixture) => { - fixture.plan.host_soft_timeout_seconds = 3300; - } - }, - { - name: "backend configured timeout", - code: "CAMPAIGN_TIMEOUT_CONFIG_MISMATCH", - mutate: (fixture) => { - fixture.backend.configured_timeout_seconds = 3300; - } - }, - { - name: "reserve-subtracted command timeout", - code: "CAMPAIGN_TIMEOUT_COMMAND_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace("--timeout 3600", "--timeout 3300"); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "duplicate timeout flag", - code: "CAMPAIGN_TIMEOUT_COMMAND_INVALID", - mutate: (fixture) => { - const command = `${fixture.backend.exact_command} --timeout 3600`; - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "missing timeout flag", - code: "CAMPAIGN_TIMEOUT_COMMAND_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace("--timeout 3600 ", ""); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "missing GNU timeout wrapper", - code: "CAMPAIGN_TIMEOUT_HOST_WRAPPER_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace( - "timeout --preserve-status --signal=INT --kill-after=300s 3600s ", - "" - ); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "wrong GNU timeout soft deadline", - code: "CAMPAIGN_TIMEOUT_HOST_WRAPPER_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace("300s 3600s recon", "300s 3300s recon"); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "wrong host force-kill grace", - code: "CAMPAIGN_TIMEOUT_HOST_GRACE_MISMATCH", - mutate: (fixture) => { - fixture.plan.host_force_kill_grace_seconds = 30; - const command = fixture.backend.exact_command.replace("--kill-after=300s", "--kill-after=30s"); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "foreground wrapper", - code: "CAMPAIGN_TIMEOUT_HOST_WRAPPER_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace("--preserve-status", "--preserve-status --foreground"); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "bounded default test limit", - code: "CAMPAIGN_TIMEOUT_COMMAND_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace( - `--test-limit ${RECON_TIMEOUT_TEST_LIMIT}`, - "--test-limit 50000" - ); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "one-step stateful sequence in plan", - code: "CAMPAIGN_SEQUENCE_LENGTH_MISMATCH", - mutate: (fixture) => { - fixture.plan.recon_sequence_length = 1; - } - }, - { - name: "one-step stateful sequence in result", - code: "CAMPAIGN_SEQUENCE_LENGTH_MISMATCH", - mutate: (fixture) => { - fixture.backend.sequence_length = 1; - } - }, - { - name: "one-step stateful sequence in summary", - code: "CAMPAIGN_SEQUENCE_LENGTH_MISMATCH", - mutate: (fixture) => { - fixture.summary.sequence_length = 1; - } - }, - { - name: "one-step stateful sequence command", - code: "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace("--seq-len 100", "--seq-len 1"); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "missing stateful sequence command flag", - code: "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace(" --seq-len 100", ""); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "stateful sequence flag before a compound Recon command", - code: "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - mutate: (fixture) => { - const command = `echo --seq-len 100 >/dev/null && ${fixture.backend.exact_command.replace(" --seq-len 100", "")}`; - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - ...["&&", "||", ";", "|", "&"].map((operator) => ({ - name: `attached ${operator} compound command`, - code: "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - mutate: (fixture: CampaignTimeoutFixture) => { - const command = `${fixture.backend.exact_command}${operator}recon fuzz . --config smoke.yaml`; - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - })), - { - name: "newline-delimited evidence command", - code: "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - mutate: (fixture) => { - const command = `${fixture.backend.exact_command.replace(" --seq-len 100", "")}\necho --seq-len 100`; - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "non-executed Recon text passed to another command", - code: "CAMPAIGN_TIMEOUT_HOST_WRAPPER_INVALID", - mutate: (fixture) => { - const command = `echo ${fixture.backend.exact_command}`; - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "commented stateful sequence flag", - code: "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace(" --seq-len 100", " # --seq-len 100"); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "quoted stateful sequence text", - code: "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace("--seq-len 100", "'--seq-len 100'"); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "duplicate stateful sequence flags", - code: "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - mutate: (fixture) => { - const command = `${fixture.backend.exact_command} --seq-len 100`; - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "stateful sequence flag after the option terminator", - code: "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - mutate: (fixture) => { - const command = fixture.backend.exact_command.replace("--seq-len 100", "-- --seq-len 100"); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - } - }, - { - name: "plan test limit", - code: "CAMPAIGN_TIMEOUT_TEST_LIMIT_MISMATCH", - mutate: (fixture) => { - fixture.plan.recon_test_limit = "50000"; - } - }, - { - name: "non-positive host force-kill grace", - code: "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - mutate: (fixture) => { - fixture.plan.host_force_kill_grace_seconds = 0; - } - }, - { - name: "non-positive artifact finalization reserve", - code: "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - mutate: (fixture) => { - fixture.plan.artifact_finalization_reserve_seconds = 0; - } - }, - { - name: "positive but forged artifact finalization reserve", - code: "CAMPAIGN_TIMEOUT_FINALIZATION_RESERVE_MISMATCH", - mutate: (fixture) => { - fixture.plan.artifact_finalization_reserve_seconds = 299; - } - }, - { - name: "backend command differs from plan", - code: "CAMPAIGN_TIMEOUT_COMMAND_MISMATCH", - mutate: (fixture) => { - fixture.backend.exact_command = `${fixture.backend.exact_command} --quiet`; - } - }, - { - name: "backend start differs from plan", - code: "CAMPAIGN_TIMEOUT_START_MISMATCH", - mutate: (fixture) => { - fixture.backend.start_timestamp = "2026-08-11T00:00:01.000Z"; - } - }, - { - name: "unknown termination reason", - code: "CAMPAIGN_TIMEOUT_EVIDENCE_INVALID", - mutate: (fixture) => { - fixture.backend.termination_reason = "unknown"; - } +const withReplacedCampaignCommand = (search: string, replacement: string) => (fixture: CampaignTimeoutFixture) => + withCampaignCommand(fixture, fixture.backend.exact_command.replace(search, replacement)); +const forgedCampaignTimeoutCases: Array<{ + name: string; + path: string; + mutate: (fixture: CampaignTimeoutFixture) => void; +}> = [ + { + name: "plan configured timeout", + path: "$.campaign_plan_ref#configured_fuzzer_timeout_seconds", + mutate: (fixture) => { + fixture.plan.configured_fuzzer_timeout_seconds = 3300; } - ]; + }, + { + name: "Recon internal timeout", + path: "$.campaign_plan_ref#recon_internal_timeout_seconds", + mutate: (fixture) => { + fixture.plan.recon_internal_timeout_seconds = 3300; + } + }, + { + name: "host soft timeout", + path: "$.campaign_plan_ref#host_soft_timeout_seconds", + mutate: (fixture) => { + fixture.plan.host_soft_timeout_seconds = 3300; + } + }, + { + name: "backend configured timeout", + path: "$.configured_timeout_seconds", + mutate: (fixture) => { + fixture.backend.configured_timeout_seconds = 3300; + } + }, + { + name: "wrong host force-kill grace", + path: "$.campaign_plan_ref#host_force_kill_grace_seconds", + mutate: (fixture) => { + fixture.plan.host_force_kill_grace_seconds = 30; + } + }, + { + name: "forged artifact finalization reserve", + path: "$.campaign_plan_ref#artifact_finalization_reserve_seconds", + mutate: (fixture) => { + fixture.plan.artifact_finalization_reserve_seconds = 299; + } + }, + { + name: "plan test limit", + path: "$.campaign_plan_ref#recon_test_limit", + mutate: (fixture) => { + fixture.plan.recon_test_limit = "50000"; + } + }, + { + name: "reserve-subtracted command timeout", + path: "$.exact_command", + mutate: withReplacedCampaignCommand("--timeout 3600", "--timeout 3300") + }, + { + name: "duplicate timeout flag", + path: "$.exact_command", + mutate: withReplacedCampaignCommand("--timeout 3600", "--timeout 3600 --timeout 3600") + }, + { name: "missing timeout flag", path: "$.exact_command", mutate: withReplacedCampaignCommand("--timeout 3600 ", "") }, + { + name: "missing GNU timeout wrapper", + path: "$.exact_command", + mutate: withReplacedCampaignCommand("timeout --preserve-status --signal=INT --kill-after=300s 3600s ", "") + }, + { + name: "wrong GNU timeout soft deadline", + path: "$.exact_command", + mutate: withReplacedCampaignCommand("300s 3600s recon", "300s 3300s recon") + }, + { + name: "foreground wrapper", + path: "$.exact_command", + mutate: withReplacedCampaignCommand("--preserve-status", "--preserve-status --foreground") + }, + { + name: "bounded default test limit", + path: "$.exact_command", + mutate: withReplacedCampaignCommand(`--test-limit ${RECON_TIMEOUT_TEST_LIMIT}`, "--test-limit 50000") + }, + { + name: "one-step stateful sequence command", + path: "$.exact_command", + mutate: withReplacedCampaignCommand("--seq-len 100", "--seq-len 1") + }, + { + name: "missing stateful sequence command flag", + path: "$.exact_command", + mutate: withReplacedCampaignCommand(" --seq-len 100", "") + }, + { + name: "duplicate stateful sequence flags", + path: "$.exact_command", + mutate: withReplacedCampaignCommand("--seq-len 100", "--seq-len 100 --seq-len 100") + }, + { + name: "one-step stateful sequence in plan", + path: "$.campaign_plan_ref#recon_sequence_length", + mutate: (fixture) => { + fixture.plan.recon_sequence_length = 1; + } + }, + { + name: "one-step stateful sequence in result", + path: "$.sequence_length", + mutate: (fixture) => { + fixture.backend.sequence_length = 1; + } + }, + { + name: "one-step stateful sequence in summary", + path: "$.campaign_summary_ref#sequence_length", + mutate: (fixture) => { + fixture.summary.sequence_length = 1; + } + }, + { + name: "backend command differs from plan", + path: "$.exact_command", + mutate: (fixture) => { + fixture.backend.exact_command = `${fixture.backend.exact_command} --quiet`; + } + }, + { + name: "backend start differs from plan", + path: "$.start_timestamp", + mutate: (fixture) => { + fixture.backend.start_timestamp = "2026-08-11T00:00:01.000Z"; + } + }, + ...(["fuzzing_deadline_utc", "force_kill_deadline_utc", "final_artifact_deadline_utc"] as const).map((field) => ({ + name: `deadline arithmetic ${field}`, + path: `$.campaign_plan_ref#${field}`, + mutate: (fixture: CampaignTimeoutFixture) => { + fixture.plan[field] = "2026-08-11T01:00:02.000Z"; + } + })), + { + name: "summary outcome", + path: "$.campaign_summary_ref#outcome", + mutate: (fixture) => { + fixture.summary.outcome = "partial"; + } + } +]; - for (const entry of cases) { +test("current campaign timeout gate rejects forged budgets, Recon flags, and sequence lengths", () => { + for (const entry of forgedCampaignTimeoutCases) { const result = runCampaignTimeoutGate(entry.mutate); assert.equal(result.ok, false, entry.name); assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === entry.code), + timeoutEvidenceIssuePaths(result).includes(entry.path), `${entry.name}: ${JSON.stringify(result.diagnostics)}` ); } }); -test("current campaign timeout gate accepts safe output redirections", () => { - for (const redirection of ["> /tmp/recon.log 2>&1", ">> /tmp/recon.log", "1> /tmp/recon.log", "&> /tmp/recon.log"]) { - const result = runCampaignTimeoutGate((fixture) => { - const command = `${fixture.backend.exact_command} ${redirection}`; - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - fixture.plan.command_plan[0]!.command = command; - }); - assert.equal(result.ok, true, `${redirection}: ${JSON.stringify(result.diagnostics)}`); +// The recorded command is evidence the host compares, never a string it runs. +// The host used to re-tokenize it with its own shell grammar and reject framing +// the verifier accepted, after the whole fuzzing budget had been spent. +test("current campaign timeout gate accepts ordinary shell framing around the recorded Recon command", () => { + const templateCommand = + "timeout --preserve-status --signal=INT --kill-after=300s 3600s recon fuzz . --contract CryticTester " + + `--test-mode assertion --workers 8 --test-limit ${RECON_TIMEOUT_TEST_LIMIT} --seq-len 100 --timeout 3600 ` + + "--corpus-dir echidna --recon-corpus-dir recon-corpus"; + for (const command of [ + `cd /workspace && ${templateCommand}`, + `${templateCommand} 2>&1 | tee backends/recon-fuzzer/run.log`, + `${templateCommand}; echo "exit=$?"`, + // The #582 shape: environment assignment before the wrapper and a redirected log. + `FOUNDRY_CACHE_PATH=/tmp/foundry-cache ${templateCommand} > /tmp/recon-fuzzer-attempt2.log 2>&1`, + ...["> /tmp/recon.log 2>&1", ">> /tmp/recon.log", "1> /tmp/recon.log", "&> /tmp/recon.log"].map( + (redirection) => `${templateCommand} ${redirection}` + ) + ]) { + const result = runCampaignTimeoutGate((fixture) => withCampaignCommand(fixture, command)); + assert.equal(result.ok, true, `${command}: ${JSON.stringify(result.diagnostics)}`); } }); -test("current campaign timeout gate preserves the specific missing-flag diagnostic with output redirection", () => { +// The result schema makes sequence_length optional and the campaign prompt does +// not ask for it; the deleted host copy required it after the budget was spent. +test("current campaign timeout gate accepts a result that omits the optional sequence_length", () => { const result = runCampaignTimeoutGate((fixture) => { - const command = `${fixture.backend.exact_command.replace(" --timeout 3600", "")} > /tmp/recon.log 2>&1`; - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - fixture.plan.command_plan[0]!.command = command; + delete fixture.backend.sequence_length; }); - const commandDiagnostics = result.diagnostics.filter((diagnostic) => - [ - "CAMPAIGN_TIMEOUT_COMMAND_INVALID", - "CAMPAIGN_SEQUENCE_LENGTH_COMMAND_INVALID", - "CAMPAIGN_TIMEOUT_HOST_WRAPPER_INVALID" - ].includes(diagnostic.code) - ); - assert.deepEqual( - commandDiagnostics.map((diagnostic) => [diagnostic.code, diagnostic.message]), - [["CAMPAIGN_TIMEOUT_COMMAND_INVALID", "Recon command must contain exactly one --timeout 3600 flag"]] - ); + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); }); -test("current campaign timeout gate does not treat non-shell whitespace as an argument boundary", () => { - for (const whitespace of ["\f", "\v", "\u00a0"]) { - const result = runCampaignTimeoutGate((fixture) => { - const command = fixture.backend.exact_command.replace("--timeout 3600", `--timeout${whitespace}3600`); - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - fixture.plan.command_plan[0]!.command = command; - }); - assert.equal(result.ok, false, JSON.stringify(result.diagnostics)); - assert.ok( - result.diagnostics.some( +test("current campaign timeout gate names only the missing flag when the command also redirects output", () => { + const result = runCampaignTimeoutGate((fixture) => + withCampaignCommand( + fixture, + `${fixture.backend.exact_command.replace(" --timeout 3600", "")} > /tmp/recon.log 2>&1` + ) + ); + assert.deepEqual( + result.diagnostics + .filter( (diagnostic) => - diagnostic.code === "CAMPAIGN_TIMEOUT_COMMAND_INVALID" && diagnostic.message.includes("--timeout 3600") - ), - JSON.stringify(result.diagnostics) - ); - } -}); - -test("current campaign timeout gate rejects shell operators after output redirection", () => { - for (const operator of [";", "|", "&&", "&", "`", "$(echo unsafe)", "<"]) { - const result = runCampaignTimeoutGate((fixture) => { - const command = `${fixture.backend.exact_command} > /tmp/recon.log 2>&1${operator} echo unsafe`; - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - }); - assert.equal(result.ok, false, operator); - assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_COMMAND_INVALID"), - `${operator}: ${JSON.stringify(result.diagnostics)}` - ); - } -}); - -test("current campaign timeout gate accepts the reported #582 command shape and rejects hostile redirect operands", () => { - const reportedCommand = - "FOUNDRY_CACHE_PATH=/tmp/foundry-cache timeout --preserve-status --signal=INT --kill-after=300s 3600s recon fuzz . " + - "--contract CryticTester --test-mode assertion --workers 32 " + - `--test-limit ${RECON_TIMEOUT_TEST_LIMIT} --seq-len 100 --timeout 3600 ` + - "--corpus-dir /tmp/corpus --recon-corpus-dir /tmp/recon-corpus --repro /tmp/repro.t.sol > /tmp/recon-fuzzer-attempt2.log 2>&1"; - const withCommand = (command: string) => { - const result = runCampaignTimeoutGate((fixture) => { - fixture.backend.exact_command = command; - fixture.plan.backend.exact_shell_escaped_command = command; - fixture.plan.command_plan[0]!.command = command; - }); - return result; - }; - - assert.equal(withCommand(reportedCommand).ok, true); - for (const redirect of [ - '> "$(touch /tmp/recon-pwned)"', - '> "`touch /tmp/recon-pwned`"', - "> \"${X:=$'$(touch /tmp/recon-pwned)'}\"", - "> \"$'\\x24\\x28touch /tmp/recon-pwned\\x29'\"", - '> "$RECON_REDIRECT_TARGET"', - ">(touch /tmp/recon-pwned)", - "2&> /tmp/recon-pwned", - "2>&1foo" - ]) { - const result = withCommand(`${reportedCommand.slice(0, reportedCommand.indexOf(" > "))} ${redirect}`); - assert.equal(result.ok, false, redirect); - assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_COMMAND_INVALID"), - `${redirect}: ${JSON.stringify(result.diagnostics)}` - ); - } -}); - -test("current campaign timeout gate verifies deadline arithmetic", () => { - for (const field of ["fuzzing_deadline_utc", "force_kill_deadline_utc", "final_artifact_deadline_utc"] as const) { - const result = runCampaignTimeoutGate((fixture) => { - fixture.plan[field] = "2026-08-11T01:00:02.000Z"; - }); - assert.equal(result.ok, false, field); - assert.ok( - result.diagnostics.some( - (diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_DEADLINE_MISMATCH" && diagnostic.path?.endsWith(field) - ), - `${field}: ${JSON.stringify(result.diagnostics)}` - ); - } + diagnostic.details?.gate === "property-campaign-timeout-evidence" && + diagnostic.path?.endsWith("#$.exact_command") + ) + .map((diagnostic) => diagnostic.message), + [ + "Semantic gate property-campaign-timeout-evidence failed: Recon command must contain exactly one --timeout 3600 flag" + ] + ); }); test("current campaign timeout gate derives early-exit outcome from recorded duration and usability", () => { @@ -10917,10 +11312,10 @@ test("current campaign timeout gate derives early-exit outcome from recorded dur fixture.backend.end_timestamp = "2026-08-11T00:30:00.000Z"; }); assert.equal(falseComplete.ok, false); - for (const code of ["CAMPAIGN_TIMEOUT_DURATION_MISMATCH", "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH"]) { + for (const issuePath of ["$.end_timestamp", "$.campaign_outcome"]) { assert.ok( - falseComplete.diagnostics.some((diagnostic) => diagnostic.code === code), - `${code}: ${JSON.stringify(falseComplete.diagnostics)}` + timeoutEvidenceIssuePaths(falseComplete).includes(issuePath), + `${issuePath}: ${JSON.stringify(falseComplete.diagnostics)}` ); } @@ -10930,26 +11325,20 @@ test("current campaign timeout gate derives early-exit outcome from recorded dur fixture.summary.outcome = "partial"; }); assert.equal(earlyConfiguredPartial.ok, false); - assert.ok( - earlyConfiguredPartial.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_DURATION_MISMATCH") - ); + assert.ok(timeoutEvidenceIssuePaths(earlyConfiguredPartial).includes("$.end_timestamp")); const fullConfiguredPartial = runCampaignTimeoutGate((fixture) => { fixture.backend.campaign_outcome = "partial"; fixture.summary.outcome = "partial"; }); assert.equal(fullConfiguredPartial.ok, false); - assert.ok( - fullConfiguredPartial.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH") - ); + assert.ok(timeoutEvidenceIssuePaths(fullConfiguredPartial).includes("$.campaign_outcome")); const fullDurationProcessExit = runCampaignTimeoutGate((fixture) => { fixture.backend.termination_reason = "process-exit"; }); assert.equal(fullDurationProcessExit.ok, false); - assert.ok( - fullDurationProcessExit.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH") - ); + assert.ok(timeoutEvidenceIssuePaths(fullDurationProcessExit).includes("$.campaign_outcome")); const falsePartialWithoutResults = runCampaignTimeoutGate((fixture) => { fixture.backend.end_timestamp = "2026-08-11T00:00:01.000Z"; @@ -10959,9 +11348,7 @@ test("current campaign timeout gate derives early-exit outcome from recorded dur fixture.summary.outcome = "partial"; }); assert.equal(falsePartialWithoutResults.ok, false); - assert.ok( - falsePartialWithoutResults.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_OUTCOME_MISMATCH") - ); + assert.ok(timeoutEvidenceIssuePaths(falsePartialWithoutResults).includes("$.campaign_outcome")); const truthfulBlocked = runCampaignTimeoutGate((fixture) => { fixture.backend.end_timestamp = "2026-08-11T00:00:01.000Z"; @@ -10980,15 +11367,13 @@ test("current campaign timeout gate classifies termination after the force-kill fixture.summary.outcome = "partial"; }); assert.equal(prematureForceKill.ok, false); - assert.ok( - prematureForceKill.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_FORCE_KILL_MISMATCH") - ); + assert.ok(timeoutEvidenceIssuePaths(prematureForceKill).includes("$.termination_reason")); const falseComplete = runCampaignTimeoutGate((fixture) => { fixture.backend.end_timestamp = "2026-08-11T01:05:06.000Z"; }); assert.equal(falseComplete.ok, false); - assert.ok(falseComplete.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_FORCE_KILL_MISMATCH")); + assert.ok(timeoutEvidenceIssuePaths(falseComplete).includes("$.termination_reason")); const truthfulPartial = runCampaignTimeoutGate((fixture) => { fixture.backend.end_timestamp = "2026-08-11T01:05:06.000Z"; @@ -11008,14 +11393,6 @@ test("current campaign timeout gate classifies termination after the force-kill assert.equal(truthfulBlocked.ok, true, JSON.stringify(truthfulBlocked.diagnostics)); }); -test("current campaign timeout gate cross-checks backend and summary outcomes", () => { - const result = runCampaignTimeoutGate((fixture) => { - fixture.summary.outcome = "partial"; - }); - assert.equal(result.ok, false); - assert.ok(result.diagnostics.some((diagnostic) => diagnostic.code === "CAMPAIGN_TIMEOUT_SUMMARY_MISMATCH")); -}); - test("campaign gate accepts many counterexamples of one property deduplicated into one finding", () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-campaign-dedup" }); campaignPropertyCatalog(layout, ["property-1"]); @@ -11048,11 +11425,7 @@ test("campaign gate accepts many counterexamples of one property deduplicated in }; const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); - assert.deepEqual( - result.diagnostics.filter((diagnostic) => diagnostic.source === "property-provenance"), - [] - ); - assert.equal(result.ok, true); + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); }); test("current campaign gate accepts the exact R55 partition: 29 counterexamples, two findings", () => { @@ -11096,11 +11469,7 @@ test("current campaign gate accepts the exact R55 partition: 29 counterexamples, }; const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); - assert.deepEqual( - result.diagnostics.filter((diagnostic) => diagnostic.source === "property-provenance"), - [] - ); - assert.equal(result.ok, true); + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); }); test("current campaign gate requires partition metadata and accepts an explicit complete partition", () => { @@ -11131,16 +11500,9 @@ test("current campaign gate requires partition metadata and accepts an explicit }; const current = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(current.ok, false); - assert.ok(current.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PARTITION_REQUIRED")); - assert.equal( - current.diagnostics.filter((diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PARTITION_UNCLAIMED").length, - 2 - ); - assert.ok( - current.diagnostics - .filter((diagnostic) => diagnostic.code.startsWith("PROPERTY_CAMPAIGN_PARTITION_")) - .every((diagnostic) => diagnostic.severity === "error") - ); + const currentPaths = gateIssuePaths(current, "property-campaign-context-joins"); + assert.ok(currentPaths.includes("$.findings_ref#0"), JSON.stringify(current.diagnostics)); + assert.equal(currentPaths.filter((issuePath) => issuePath === "$.failures").length, 2); const completePartition = accountedCampaignFinding("failure-1", ["property-1"], ["failure-1", "failure-2"]); writeArtifact(layout, campaignId, "findings.json", JSON.stringify([completePartition])); @@ -11173,14 +11535,14 @@ test("campaign partition rejects unknown contributions and per-finding count mis const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); - for (const code of [ - "PROPERTY_CAMPAIGN_PARTITION_REFERENCE_UNKNOWN", - "PROPERTY_CAMPAIGN_PARTITION_COUNT_MISMATCH", - "PROPERTY_CAMPAIGN_PARTITION_UNCLAIMED" + for (const issuePath of [ + "$.findings_ref#0.contributing_backend_failures[0]", + "$.findings_ref#0.deduplication.pre_dedup_count", + "$.failures" ]) { assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === code), - code + gateIssuePaths(result, "property-campaign-context-joins").includes(issuePath), + `${issuePath}: ${JSON.stringify(result.diagnostics)}` ); } }); @@ -11221,9 +11583,14 @@ test("campaign partition rejects duplicate claims and property subset mismatches const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); - assert.ok(result.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PARTITION_DUPLICATE")); + const issues = campaignJoinIssues(result); assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PARTITION_PROPERTY_MISMATCH") + issues.some((entry) => entry.includes('"failure-2" must be claimed by exactly one finding')), + JSON.stringify(issues) + ); + assert.ok( + issues.some((entry) => entry.startsWith("$.findings_ref#0.property_ids:")), + JSON.stringify(issues) ); }); @@ -11263,10 +11630,9 @@ test("campaign partition requires each finding ID to represent one of its contri const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); - assert.equal( - result.diagnostics.filter((diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PARTITION_REPRESENTATIVE_MISMATCH") - .length, - 2 + assert.deepEqual( + gateIssuePaths(result, "property-campaign-context-joins").filter((issuePath) => issuePath.endsWith(".id")), + ["$.findings_ref#0.id", "$.findings_ref#1.id"] ); }); @@ -11306,10 +11672,11 @@ test("campaign partition requires a finding's properties to equal its contributi const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); - assert.equal( - result.diagnostics.filter((diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PARTITION_PROPERTY_MISMATCH") - .length, - 1 + assert.deepEqual( + gateIssuePaths(result, "property-campaign-context-joins").filter((issuePath) => + issuePath.endsWith(".property_ids") + ), + ["$.findings_ref#0.property_ids"] ); }); @@ -11352,9 +11719,7 @@ test("campaign partition binds finding backend provenance to its exact contribut const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); - assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PARTITION_BACKEND_MISMATCH") - ); + assert.deepEqual(gateIssuePaths(result, "findings-campaign-provenance-coherence"), ["$[0]"]); }); test("campaign contributions bind raw_result_ref to the authenticated campaign artifact", () => { @@ -11385,25 +11750,83 @@ test("campaign contributions bind raw_result_ref to the authenticated campaign a ) ]) ); - writeCampaignSummary(layout, campaignId, 1, 1); - const node = { - ...currentCampaignNode(["recon-fuzzer-results.json", "findings.json"]), + writeCampaignSummary(layout, campaignId, 1, 1); + const node = { + ...currentCampaignNode(["recon-fuzzer-results.json", "findings.json"]), + id: campaignId, + logical_id: campaignId + }; + + const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); + assert.equal(result.ok, false); + assert.deepEqual(gateIssuePaths(result, "property-campaign-context-joins"), [ + "$.findings_ref#0.contributing_backend_failures[0].raw_result_ref" + ]); +}); + +// The registry gate requires raw_result_ref to be the declared relative path; +// the deleted host copy required its basename, so no value satisfied both. +test("campaign contributions name a nested campaign result by its declared relative path", () => { + const layout = createRunLayout({ + projectRoot: tempProject(), + runId: "run-nested-campaign-raw-result-ref", + resolvedConfigToml: '[invariants]\ninvariant_testing_fuzzer_timeout = "1h"\n' + }); + campaignPropertyCatalog(layout, ["property-1"]); + const campaignId = "custom-fuzz-stage"; + const campaignOutput = boundOutput("custom/results.json", "ultrafuzz/property-campaign@3", true); + const findingsOutput = boundOutput("custom/candidates.json", "ultrafuzz/findings@2"); + const campaignPlanOutput = boundOutput("custom/plan.json", "ultrafuzz/invariant-campaign-plan@2"); + const campaignSummaryOutput = boundOutput("custom/summary.json", "ultrafuzz/campaign-summary@2"); + const campaign = { + ...currentCampaign(["property-1"], [{ id: "failure-1", property_ids: ["property-1"] }]), + campaign_plan_ref: campaignPlanOutput.path, + findings_ref: findingsOutput.path, + campaign_summary_ref: campaignSummaryOutput.path + }; + writeDeclaredArtifactNode( + layout, + campaignId, + [campaignOutput, findingsOutput, campaignPlanOutput, campaignSummaryOutput], + { + [campaignOutput.path]: JSON.stringify(campaign), + [findingsOutput.path]: JSON.stringify([ + accountedCampaignFinding( + "failure-1", + ["property-1"], + [{ fuzzer_backend: "recon", failure_id: "failure-1", raw_result_ref: campaignOutput.path }] + ) + ]), + [campaignPlanOutput.path]: JSON.stringify(currentCampaignPlan()), + [campaignSummaryOutput.path]: JSON.stringify({ + schema_version: "ultrafuzz.campaign-summary.v2", + outcome: "partial", + sequence_length: 100, + implemented_property_suite_refs: ["implemented-properties.json"], + campaign_plan_ref: campaignPlanOutput.path, + backend_results: [{ fuzzer_backend: "recon", status: "partial", result_ref: campaignOutput.path }], + finding_refs: ["failure-1"], + reproducer_refs: [ + { finding_id: "failure-1", path: `${campaignFixturePaths.reproducers}/failure-1.t.sol`, blocker: null } + ], + failure_counts: { pre_deduplication: 1, post_deduplication: 1 } + }) + } + ); + const node: PlannedGraphNode = { + ...plannedNode([]), id: campaignId, - logical_id: campaignId + logical_id: campaignId, + display_name: campaignId, + artifact_dir: `artifacts/${campaignId}`, + depends_on: ["stateful-invariant-implement-properties"], + timeout_seconds: 7200, + outputs: [campaignOutput, findingsOutput, campaignPlanOutput, campaignSummaryOutput] }; - const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); - assert.equal(result.ok, false); - assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PARTITION_RAW_RESULT_MISMATCH") - ); - assert.ok( - result.diagnostics.some( - (diagnostic) => - diagnostic.code === "ARTIFACT_SEMANTIC_GATE_FAILED" && - diagnostic.details?.gate === "property-campaign-context-joins" - ) - ); + const result = verifyRequiredArtifactsForAttempt(layout, node, node.id); + + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); }); test("campaign failures and findings cannot name selected but non-implemented properties", () => { @@ -11453,17 +11876,10 @@ test("campaign failures and findings cannot name selected but non-implemented pr const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); - assert.ok( - result.diagnostics.filter((diagnostic) => diagnostic.code === "PROPERTY_IMPLEMENTATION_REFERENCE_INVALID").length >= - 2 - ); - assert.ok( - result.diagnostics.some( - (diagnostic) => - diagnostic.code === "ARTIFACT_SEMANTIC_GATE_FAILED" && - diagnostic.details?.gate === "property-campaign-context-joins" - ) - ); + const issuePaths = gateIssuePaths(result, "property-campaign-context-joins"); + for (const issuePath of ["$.failures[0].property_ids[0]", "$.findings_ref#0.property_ids[0]"]) { + assert.ok(issuePaths.includes(issuePath), `${issuePath}: ${JSON.stringify(result.diagnostics)}`); + } }); test("campaign contract rejects v2 bytes without converting or rewriting them", () => { @@ -11543,12 +11959,10 @@ test("campaign gate conditionally reconciles the R55 summary failure counts with diagnostic.details?.gate === "campaign-summary-count-coupling" ) ); - assert.deepEqual( - mismatched.diagnostics - .filter((diagnostic) => diagnostic.code === "CAMPAIGN_SUMMARY_FAILURE_COUNT_MISMATCH") - .map((diagnostic) => diagnostic.path?.split(".").at(-1)), - ["pre_deduplication", "post_deduplication"] - ); + assert.deepEqual(gateIssuePaths(mismatched, "campaign-summary-count-coupling"), [ + "$.failure_counts.pre_deduplication", + "$.failure_counts.post_deduplication" + ]); const legacyPath = writeArtifact(layout, campaignId, "campaign-summary.json", JSON.stringify({ outcome: "partial" })); const legacyBytes = fs.readFileSync(legacyPath); @@ -11579,151 +11993,6 @@ test("campaign gate conditionally reconciles the R55 summary failure counts with assert.deepEqual(fs.readFileSync(incompletePath), incompleteBytes); }); -test("campaign gate still rejects a property-derived failure no finding covers", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-campaign-uncovered" }); - campaignPropertyCatalog(layout, ["property-1", "property-2"]); - const campaignId = "stateful-invariant-campaign"; - writeArtifact( - layout, - campaignId, - "recon-fuzzer-results.json", - JSON.stringify( - currentCampaign( - ["property-1", "property-2"], - [ - { id: "failure-1", property_ids: ["property-1"] }, - { id: "failure-2", property_ids: ["property-2"] } - ] - ) - ) - ); - writeArtifact(layout, campaignId, "findings.json", JSON.stringify([campaignFinding("failure-1", ["property-1"])])); - writeCampaignSummary(layout, campaignId, 2, 1); - const node = { - ...currentCampaignNode(["recon-fuzzer-results.json", "findings.json"]), - id: campaignId, - logical_id: campaignId - }; - - const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); - assert.equal(result.ok, false); - const missing = result.diagnostics.find((diagnostic) => diagnostic.code === "PROPERTY_FINDING_REFERENCE_MISSING"); - assert.ok(missing); - assert.match(missing?.path ?? "", /failures\[1\]/u); -}); - -test("campaign gate does not let an unrelated finding cover a campaign failure", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-campaign-unrelated-coverage" }); - campaignPropertyCatalog(layout, ["property-1"]); - const campaignId = "stateful-invariant-campaign"; - writeArtifact( - layout, - campaignId, - "recon-fuzzer-results.json", - JSON.stringify(currentCampaign(["property-1"], [{ id: "failure-1", property_ids: ["property-1"] }])) - ); - writeArtifact( - layout, - campaignId, - "findings.json", - JSON.stringify([campaignFinding("unrelated-finding", ["property-1"])]) - ); - writeCampaignSummary(layout, campaignId, 1, 1); - const node = { - ...currentCampaignNode(["recon-fuzzer-results.json", "findings.json"]), - id: campaignId, - logical_id: campaignId - }; - - const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); - assert.equal(result.ok, false); - assert.ok(result.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_FINDING_REFERENCE_MISSING")); -}); - -test("campaign gate names only the genuinely uncovered property of a partially covered failure", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-campaign-partial" }); - campaignPropertyCatalog(layout, ["property-1", "property-2"]); - const campaignId = "stateful-invariant-campaign"; - writeArtifact( - layout, - campaignId, - "recon-fuzzer-results.json", - JSON.stringify( - currentCampaign( - ["property-1", "property-2"], - [ - { id: "failure-1", property_ids: ["property-1", "property-2"] }, - { id: "failure-2", property_ids: ["property-1"] } - ] - ) - ) - ); - writeArtifact(layout, campaignId, "findings.json", JSON.stringify([campaignFinding("failure-2", ["property-1"])])); - writeCampaignSummary(layout, campaignId, 2, 1); - const node = { - ...currentCampaignNode(["recon-fuzzer-results.json", "findings.json"]), - id: campaignId, - logical_id: campaignId - }; - - const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); - assert.equal(result.ok, false); - const missing = result.diagnostics.find((diagnostic) => diagnostic.code === "PROPERTY_FINDING_REFERENCE_MISSING"); - assert.ok(missing); - assert.match(missing?.message ?? "", /property-2/u); - // property-1 is covered by the finding, so naming it would send the retry - // after an artifact that is already correct. - assert.doesNotMatch(missing?.message ?? "", /property-1/u); -}); - -test("campaign gate keeps flagging ambiguous and mismatched same-ID findings", () => { - const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-campaign-ambiguous" }); - campaignPropertyCatalog(layout, ["property-1"]); - const campaignId = "stateful-invariant-campaign"; - writeArtifact( - layout, - campaignId, - "recon-fuzzer-results.json", - JSON.stringify( - currentCampaign( - ["property-1"], - [ - { id: "failure-1", property_ids: ["property-1"] }, - { id: "failure-2", property_ids: ["property-1"] } - ] - ) - ) - ); - writeArtifact( - layout, - campaignId, - "findings.json", - JSON.stringify([campaignFinding("failure-1", ["property-1"]), campaignFinding("failure-1", ["property-1"])]) - ); - writeCampaignSummary(layout, campaignId, 2, 2); - const node = { - ...currentCampaignNode(["recon-fuzzer-results.json", "findings.json"]), - id: campaignId, - logical_id: campaignId - }; - const ambiguous = verifyRequiredArtifactsForAttempt(layout, node, campaignId); - assert.equal(ambiguous.ok, false); - assert.ok(ambiguous.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_FINDING_REFERENCE_AMBIGUOUS")); - - // A finding that claims a failure's ID must still carry that failure's properties, - // even though other failures may now be covered by a different finding. - writeArtifact( - layout, - campaignId, - "findings.json", - JSON.stringify([campaignFinding("failure-1", []), campaignFinding("failure-2", ["property-1"])]) - ); - writeCampaignSummary(layout, campaignId, 2, 2); - const mismatched = verifyRequiredArtifactsForAttempt(layout, node, campaignId); - assert.equal(mismatched.ok, false); - assert.ok(mismatched.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_FINDING_REFERENCE_MISMATCH")); -}); - test("campaign gate rejects a failure whose property combination no single finding claims", () => { const layout = createRunLayout({ projectRoot: tempProject(), runId: "run-campaign-combination" }); campaignPropertyCatalog(layout, ["property-1", "property-2"]); @@ -11745,12 +12014,11 @@ test("campaign gate rejects a failure whose property combination no single findi ) ) ); - writeArtifact( - layout, - campaignId, - "findings.json", - JSON.stringify([campaignFinding("failure-1", ["property-1"]), campaignFinding("failure-2", ["property-2"])]) - ); + const singlePropertyFindings = [ + accountedCampaignFinding("failure-1", ["property-1"], ["failure-1"]), + accountedCampaignFinding("failure-2", ["property-2"], ["failure-2"]) + ]; + writeArtifact(layout, campaignId, "findings.json", JSON.stringify(singlePropertyFindings)); writeCampaignSummary(layout, campaignId, 3, 2); const node = { ...currentCampaignNode(["recon-fuzzer-results.json", "findings.json"]), @@ -11760,9 +12028,22 @@ test("campaign gate rejects a failure whose property combination no single findi const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); - const missing = result.diagnostics.find((diagnostic) => diagnostic.code === "PROPERTY_FINDING_REFERENCE_MISSING"); - assert.ok(missing); - assert.match(missing?.message ?? "", /failure-3/u); + assert.deepEqual(campaignJoinIssues(result), [ + '$.failures: Semantic gate property-campaign-context-joins failed: Property-derived campaign failure "failure-3" must be claimed by exactly one finding' + ]); + + writeArtifact( + layout, + campaignId, + "findings.json", + JSON.stringify([ + ...singlePropertyFindings, + accountedCampaignFinding("failure-3", ["property-1", "property-2"], ["failure-3"]) + ]) + ); + writeCampaignSummary(layout, campaignId, 3, 3); + const claimed = verifyRequiredArtifactsForAttempt(layout, node, campaignId); + assert.equal(claimed.ok, true, JSON.stringify(claimed.diagnostics)); }); test("campaign gate accepts a deduplicated finding that unions the properties of the failures it covers", () => { @@ -11800,11 +12081,7 @@ test("campaign gate accepts a deduplicated finding that unions the properties of }; const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); - assert.deepEqual( - result.diagnostics.filter((diagnostic) => diagnostic.source === "property-provenance"), - [] - ); - assert.equal(result.ok, true); + assert.equal(result.ok, true, JSON.stringify(result.diagnostics)); }); test("campaign gate rejects a finding that claims a property no failure ever reported", () => { @@ -11823,7 +12100,7 @@ test("campaign gate rejects a finding that claims a property no failure ever rep layout, campaignId, "findings.json", - JSON.stringify([campaignFinding("failure-1", ["property-1", "property-2"])]) + JSON.stringify([accountedCampaignFinding("failure-1", ["property-1", "property-2"], ["failure-1"])]) ); writeCampaignSummary(layout, campaignId, 1, 1); const node = { @@ -11834,12 +12111,7 @@ test("campaign gate rejects a finding that claims a property no failure ever rep const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); - const unobserved = result.diagnostics.find( - (diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PROPERTY_UNOBSERVED" - ); - assert.ok(unobserved); - assert.match(unobserved?.message ?? "", /property-2/u); - assert.doesNotMatch(unobserved?.message ?? "", /property-1/u); + assert.deepEqual(gateIssuePaths(result, "property-campaign-context-joins"), ["$.findings_ref#0.property_ids"]); }); test("campaign gate rejects a property claim anchored to a failure that reported no property", () => { @@ -11854,7 +12126,12 @@ test("campaign gate rejects a property claim anchored to a failure that reported "recon-fuzzer-results.json", JSON.stringify(currentCampaign(["property-1"], [{ id: "failure-1", property_ids: [] }])) ); - writeArtifact(layout, campaignId, "findings.json", JSON.stringify([campaignFinding("failure-1", ["property-1"])])); + writeArtifact( + layout, + campaignId, + "findings.json", + JSON.stringify([accountedCampaignFinding("failure-1", ["property-1"], ["failure-1"])]) + ); writeCampaignSummary(layout, campaignId, 1, 1); const node = { ...currentCampaignNode(["recon-fuzzer-results.json", "findings.json"]), @@ -11865,8 +12142,10 @@ test("campaign gate rejects a property claim anchored to a failure that reported const result = verifyRequiredArtifactsForAttempt(layout, node, campaignId); assert.equal(result.ok, false); assert.ok( - result.diagnostics.some((diagnostic) => diagnostic.code === "PROPERTY_CAMPAIGN_PROPERTY_UNOBSERVED"), - `expected an unobserved-property diagnostic, got ${JSON.stringify(result.diagnostics.map((d) => d.code))}` + gateIssuePaths(result, "property-campaign-context-joins").includes( + "$.findings_ref#0.contributing_backend_failures[0]" + ), + JSON.stringify(result.diagnostics) ); }); @@ -12568,10 +12847,16 @@ test("coverage gate binds selected and unselected ranges to the trusted producti ]) { publish(evidence, `${scopedMarkdown}\n## Notes\n\n${unscopedProducerScore}\n`); const hiddenProducerScope = verifyRequiredArtifactsForAttempt(layout, node, node.id); - assert.equal(hiddenProducerScope.ok, false, unscopedProducerScore); + // Prose scores are advisory; the typed evidence and canonical section bind the result. + assert.equal( + hiddenProducerScope.ok, + true, + `${unscopedProducerScore}: ${JSON.stringify(hiddenProducerScope.diagnostics)}` + ); assert.ok( - hiddenProducerScope.diagnostics.some((diagnostic) => - /^UNSCOPED_COVERAGE_(?:FRACTION|PERCENTAGE)$/u.test(diagnostic.code) + hiddenProducerScope.diagnostics.some( + (diagnostic) => + /^UNSCOPED_COVERAGE_(?:FRACTION|PERCENTAGE)$/u.test(diagnostic.code) && diagnostic.severity === "warning" ), JSON.stringify(hiddenProducerScope.diagnostics) ); @@ -12903,13 +13188,22 @@ test("coverage gate binds selected and unselected ranges to the trusted producti JSON.stringify(contradictoryScopedFraction.diagnostics) ); - publish(evidence, `${scopedMarkdown}\n## Notes\n\nRetry 1/2 reproduced the same revert.\n`); - const ordinaryFraction = verifyRequiredArtifactsForAttempt(layout, node, node.id); - assert.equal(ordinaryFraction.ok, false); - assert.ok( - ordinaryFraction.diagnostics.some((diagnostic) => diagnostic.code === "UNSCOPED_COVERAGE_SCORE"), - JSON.stringify(ordinaryFraction.diagnostics) - ); + for (const ordinaryProse of [ + "Retry 1/2 reproduced the same revert.", + "Handlers reachable: 7/9", + "Target: 90% of Recon-selected declarations.", + "After iteration 2 we covered 41 of 57 functions." + ]) { + publish(evidence, `${scopedMarkdown}\n## Notes\n\n${ordinaryProse}\n`); + const ordinary = verifyRequiredArtifactsForAttempt(layout, node, node.id); + assert.equal(ordinary.ok, true, `${ordinaryProse}: ${JSON.stringify(ordinary.diagnostics)}`); + assert.ok( + ordinary.diagnostics.some( + (diagnostic) => diagnostic.code === "UNSCOPED_COVERAGE_SCORE" && diagnostic.severity === "warning" + ), + `${ordinaryProse}: ${JSON.stringify(ordinary.diagnostics)}` + ); + } const staleGoal = structuredClone(goal); staleGoal.current_measurement.covered_ranges = 0; @@ -13715,6 +14009,24 @@ test("coverage gate authenticates Vyper declaration boundaries in the Recon sele ); }); +/** Natural-language coverage scores are reported only as advisory warnings, never as gate errors. */ +function assertAdvisoryCoverageScore( + result: ReturnType, + code: "UNSCOPED_COVERAGE_PERCENTAGE" | "UNSCOPED_COVERAGE_FRACTION", + label: string +): void { + assert.ok( + result.diagnostics.some((diagnostic) => diagnostic.code === code && diagnostic.severity === "warning"), + `${label}: ${JSON.stringify(result.diagnostics)}` + ); + assert.ok( + result.diagnostics.every( + (diagnostic) => !diagnostic.code.startsWith("UNSCOPED_COVERAGE_") || diagnostic.severity === "warning" + ), + `${label}: ${JSON.stringify(result.diagnostics)}` + ); +} + test("final report preserves typed coverage evidence and its canonical Markdown projection", () => { const lcovArtifactPath = "reports/2026/08/coverage-input.lcov"; const layout = createRunLayout({ @@ -14094,16 +14406,12 @@ test("final report preserves typed coverage evidence and its canonical Markdown ]) { writeArtifact(layout, reportNode.id, "report.md", `${scopedMarkdown}\n## Notes\n\n${mixedScore}\n`); const mixed = verifyRequiredArtifactsForAttempt(layout, reportNode, reportNode.id); - assert.equal(mixed.ok, false, mixedScore); const normalizedMixedScore = mixedScore.normalize("NFKC").replace(/[\u2044\u2215\u29f8]/gu, "/"); - assert.ok( - mixed.diagnostics.some( - (diagnostic) => - diagnostic.code === - (/%|٪|&(?:percnt|#0*37|#x0*25);|\bpct\b\.?|\bper[ -]?cent(?:age)?\b/iu.test(normalizedMixedScore) - ? "UNSCOPED_COVERAGE_PERCENTAGE" - : "UNSCOPED_COVERAGE_FRACTION") - ), + assertAdvisoryCoverageScore( + mixed, + /%|٪|&(?:percnt|#0*37|#x0*25);|\bpct\b\.?|\bper[ -]?cent(?:age)?\b/iu.test(normalizedMixedScore) + ? "UNSCOPED_COVERAGE_PERCENTAGE" + : "UNSCOPED_COVERAGE_FRACTION", mixedScore ); } @@ -14216,11 +14524,8 @@ test("final report preserves typed coverage evidence and its canonical Markdown ]) { writeArtifact(layout, reportNode.id, "report.md", `${scopedMarkdown}\n## Notes\n\n${crossRenderedLineScope}\n`); const crossLine = verifyRequiredArtifactsForAttempt(layout, reportNode, reportNode.id); - assert.equal(crossLine.ok, false, crossRenderedLineScope); - assert.ok( - crossLine.diagnostics.some((diagnostic) => diagnostic.code === "UNSCOPED_COVERAGE_PERCENTAGE"), - `${crossRenderedLineScope}: ${JSON.stringify(crossLine.diagnostics)}` - ); + assert.equal(crossLine.ok, true, `${crossRenderedLineScope}: ${JSON.stringify(crossLine.diagnostics)}`); + assertAdvisoryCoverageScore(crossLine, "UNSCOPED_COVERAGE_PERCENTAGE", crossRenderedLineScope); } for (const implicitlyVisibleScore of [ @@ -14252,11 +14557,13 @@ test("final report preserves typed coverage evidence and its canonical Markdown ]) { writeArtifact(layout, reportNode.id, "report.md", `${scopedMarkdown}\n## Notes\n\n${implicitlyVisibleScore}\n`); const visibleAfterImplicitClose = verifyRequiredArtifactsForAttempt(layout, reportNode, reportNode.id); - assert.equal(visibleAfterImplicitClose.ok, false, implicitlyVisibleScore); - assert.ok( - visibleAfterImplicitClose.diagnostics.some((diagnostic) => diagnostic.code === "UNSCOPED_COVERAGE_PERCENTAGE"), + // Only the stylesheet cases fail; the prose score itself stays advisory. + assert.equal( + visibleAfterImplicitClose.ok, + !implicitlyVisibleScore.startsWith("