From 6fffcf40c7a05877c10b68c583d2e2e763620640 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Mon, 28 Sep 2026 22:58:29 +0000 Subject: [PATCH 1/9] fix(runtime): restate whole-run accounting in runtime report presentations The final-report agent copies a run summary that the host captures when the report task starts, and both runtime presentations (the verified terminal publication and the unchecked fallback report) republished that copy unchanged. Elapsed time, models, tokens and spend therefore excluded the report task itself and everything that finished after it started. On a local smoke run the published report showed 2,363,772 tokens, $5.64 and 20m 17s, while run.json recorded 4,591,556 tokens and $10.17 and the run finished after 36m 23s. projectTerminalReport and captureUnverifiedReport already read the final run.json and state.json, so withWholeRunSummary() restates elapsed time (run.json created_at to state.json finished_at) and models, tokens, spend and partial_pricing from accounting.cumulative. A missing, malformed or "unavailable" value keeps the agent's copy, and the helper never throws. The agent's own report.json and report.md are untouched. Unchecked reports also never received the run-root goal-search census, so they always said "Goal search coverage is unknown". They now read goal-search-coverage.json the same way the verified path does; an unreadable census still renders as unknown coverage instead of failing the report. Refs #1151 Co-Authored-By: Claude Opus 5.5 --- .../runtime/src/terminal-report-projection.ts | 51 +++++++ .../runtime/src/unverified-report-inputs.ts | 12 ++ packages/runtime/src/unverified-report.ts | 19 ++- packages/runtime/test/runtime.test.ts | 9 +- .../test/terminal-report-projection.test.ts | 131 +++++++++++++++++- .../runtime/test/unverified-report.test.ts | 56 ++++++++ packages/runtime/test/verified-output.test.ts | 9 +- 7 files changed, 278 insertions(+), 9 deletions(-) diff --git a/packages/runtime/src/terminal-report-projection.ts b/packages/runtime/src/terminal-report-projection.ts index 5f70cc42e..ebc6094cd 100644 --- a/packages/runtime/src/terminal-report-projection.ts +++ b/packages/runtime/src/terminal-report-projection.ts @@ -51,8 +51,59 @@ export function projectTerminalReport(input: TerminalReportProjectionInput): Can if (sourceRunId !== undefined && Reflect.get(reportMetadata, "source_run_id") !== sourceRunId) { throw new Error("Verified final report has a different source run identity"); } + report.run_metadata = withWholeRunSummary(reportMetadata as Record, metadata, state.finished_at); report.completion = completion; return projectCanonicalFinalReport(report, context); } throw new ReportUnavailableError("no successful report-agent output is available"); } + +/** + * The report agent copies a run summary that the host captured when the report task started, so its + * elapsed time and accounting miss the report task itself and anything that finished later. Runtime + * presentations restate them from run.json and the recorded finish time. A value those records do + * not provide keeps the agent's copy; malformed records are ignored, never thrown. + */ +export function withWholeRunSummary( + runMetadata: Record, + metadata: unknown, + finishedAt: unknown +): Record { + const summary = { ...runMetadata }; + const elapsed = elapsedTime(field(metadata, "created_at"), finishedAt); + if (elapsed !== undefined) summary.elapsed_time = elapsed; + const cumulative = field(field(metadata, "accounting"), "cumulative"); + const models = field(cumulative, "models"); + if (isNonEmptyStringList(models)) summary.models_used = [...models]; + const tokens = field(cumulative, "tokens_used"); + if (availableLabel(tokens)) summary.tokens_used = tokens; + const spend = field(cumulative, "estimated_spend"); + if (availableLabel(spend)) { + summary.estimated_spend = spend; + summary.partial_pricing = field(cumulative, "partial_pricing") === true; + } + return summary; +} + +function field(value: unknown, key: string): unknown { + return typeof value === "object" && value !== null && !Array.isArray(value) ? Reflect.get(value, key) : undefined; +} + +function isNonEmptyStringList(value: unknown): value is string[] { + return Array.isArray(value) && value.length > 0 && value.every((entry) => typeof entry === "string" && entry !== ""); +} + +function availableLabel(value: unknown): value is string { + return typeof value === "string" && value.trim() !== "" && value.trim().toLowerCase() !== "unavailable"; +} + +/** Same format as the host's report-start projection: `42.0s`, `5m 07s`, or `6h 02m`. */ +function elapsedTime(createdAt: unknown, finishedAt: unknown): string | undefined { + if (typeof createdAt !== "string" || typeof finishedAt !== "string") return undefined; + const totalSeconds = (Date.parse(finishedAt) - Date.parse(createdAt)) / 1_000; + if (!Number.isFinite(totalSeconds) || totalSeconds < 0) return undefined; + if (totalSeconds < 60) return `${totalSeconds.toFixed(1)}s`; + const totalMinutes = Math.floor(totalSeconds / 60); + if (totalMinutes < 60) return `${String(totalMinutes)}m ${String(Math.floor(totalSeconds % 60)).padStart(2, "0")}s`; + return `${String(Math.floor(totalMinutes / 60))}h ${String(totalMinutes % 60).padStart(2, "0")}m`; +} diff --git a/packages/runtime/src/unverified-report-inputs.ts b/packages/runtime/src/unverified-report-inputs.ts index ff9e12e2c..32c1850fa 100644 --- a/packages/runtime/src/unverified-report-inputs.ts +++ b/packages/runtime/src/unverified-report-inputs.ts @@ -13,6 +13,7 @@ import { reportSchema, type ReportVerification } from "@ultrafuzz/artifacts"; +import { loadGoalSearchCoverageSnapshot } from "./final-report-markdown.js"; import { ReportUnavailableError } from "./report-unavailable.js"; type JsonRecord = Record; @@ -31,6 +32,7 @@ export interface UnverifiedReportInputs { observed: ObservedReportCompletion; verification: ReportVerification; agentReport: JsonRecord; + goalSearchCoverage?: unknown; sources_sha256: string; } @@ -53,10 +55,20 @@ export function readUnverifiedReportInputs(root: string): UnverifiedReportInputs observed, verification: { status: "not-checked", reason_codes: [...reader.reasons].sort() }, agentReport, + goalSearchCoverage: readGoalSearchCoverage(root), sources_sha256: sha256Bytes(Buffer.from(JSON.stringify(reader.sources))) }; } +/** An unreadable census renders as unknown goal-search coverage instead of hiding the report. */ +function readGoalSearchCoverage(root: string): unknown { + try { + return loadGoalSearchCoverageSnapshot(root); + } catch { + return undefined; + } +} + class ReportInputReader { readonly reasons = new Set(["verification-unavailable"]); readonly sources: [string, string][] = []; diff --git a/packages/runtime/src/unverified-report.ts b/packages/runtime/src/unverified-report.ts index 466d93414..8489c9b32 100644 --- a/packages/runtime/src/unverified-report.ts +++ b/packages/runtime/src/unverified-report.ts @@ -14,6 +14,7 @@ import { } from "@ultrafuzz/artifacts"; import { projectCanonicalFinalReport } from "./final-report-markdown.js"; +import { withWholeRunSummary } from "./terminal-report-projection.js"; import { loadCurrentFinalReportSnapshot, publishTerminalReport, @@ -111,11 +112,19 @@ function publishUncheckedPresentation(root: string, file: string, bytes: Buffer) function captureUnverifiedReport(inputs: UnverifiedReportInputs): ReportSnapshot { validateSafeId(inputs.runId, "run ID"); - const projection = projectCanonicalFinalReport({ - ...inputs.agentReport, - verification: inputs.verification, - observed_completion: inputs.observed - }); + const projection = projectCanonicalFinalReport( + { + ...inputs.agentReport, + run_metadata: withWholeRunSummary( + inputs.agentReport.run_metadata as Record, + inputs.metadata, + inputs.state?.finished_at + ), + verification: inputs.verification, + observed_completion: inputs.observed + }, + { goalSearchCoverage: inputs.goalSearchCoverage } + ); const jsonBytes = Buffer.from(`${JSON.stringify(projection.report, null, 2)}\n`, "utf8"); const markdownBytes = Buffer.from(projection.markdown, "utf8"); const generation = sha256Bytes(Buffer.concat([jsonBytes, markdownBytes])); diff --git a/packages/runtime/test/runtime.test.ts b/packages/runtime/test/runtime.test.ts index 357f3466d..786d1298f 100644 --- a/packages/runtime/test/runtime.test.ts +++ b/packages/runtime/test/runtime.test.ts @@ -19526,7 +19526,14 @@ test("syncRun publishes a report after recovery of a failed report agent", async assert.equal(recoveredReport.completion?.counts.succeeded, 1); assert.equal(recoveredReport.completion?.counts.failed, 0); assert.doesNotMatch(recoveredReport.markdown, /^# Ultrafuzz report — PARTIAL/u); - assert.deepEqual(recoveredReport.json, { ...finalReport.report, completion: recoveredReport.completion }); + // The run summary restates elapsed time from run.json and state.json; all review content is the agent's. + const elapsed = (recoveredReport.json as { run_metadata: { elapsed_time: string } }).run_metadata.elapsed_time; + assert.match(elapsed, /^(?:\d+\.\ds|\d+m \d{2}s)$/u); + assert.deepEqual(recoveredReport.json, { + ...finalReport.report, + run_metadata: { ...(finalReport.report.run_metadata as Record), elapsed_time: elapsed }, + completion: recoveredReport.completion + }); }); for (const variant of ["failed-verifier", "changed-output", "exhausted-loop"] as const) { diff --git a/packages/runtime/test/terminal-report-projection.test.ts b/packages/runtime/test/terminal-report-projection.test.ts index 9f1dabcd7..43c7e5b3e 100644 --- a/packages/runtime/test/terminal-report-projection.test.ts +++ b/packages/runtime/test/terminal-report-projection.test.ts @@ -133,8 +133,14 @@ test("terminal projection preserves all verified review data and warnings for co const input = { completion: census, state, metadata: metadata(), agentReport: agent }; const original = structuredClone(input); const result = projectTerminalReport(input); - assert.deepEqual(result.report, { ...before, completion: census }); - assert.deepEqual(result, projectCanonicalFinalReport({ ...before, completion: census })); + // run.json carries no accounting here, so only the whole-run elapsed time replaces the agent's copy. + const expected = { + ...before, + run_metadata: { ...(before.run_metadata as Record), elapsed_time: "2m 00s" }, + completion: census + }; + assert.deepEqual(result.report, expected); + assert.deepEqual(result, projectCanonicalFinalReport(expected)); assert.deepEqual(input, original); assert.equal(result.markdown.startsWith("# Ultrafuzz report — PARTIAL"), partial); } @@ -190,3 +196,124 @@ test("terminal projection enforces bounded and internally consistent completion assert.deepEqual(result.report.completion, bounded); assert.match(result.markdown, /identities omitted from this bounded census: `1`/u); }); + +/** A valid run.json whose cumulative accounting already includes the report task's own usage. */ +function metadataWithAccounting(cumulative: { estimated_spend?: string; models?: string[] } = {}): RunMetadataDocument { + const summary = { + uncached_input_tokens: 9_000_000, + input_tokens: 9_000_000, + output_tokens: 3_345_678, + cache_read_tokens: 0, + cache_write_tokens: 0, + reasoning_tokens: 0, + inclusive_token_total: 12_345_678, + billable_token_total: 12_345_678, + total_tokens: 12_345_678, + tokens_used: "12,345,678", + estimated_spend: "$41.20+", + estimated_spend_usd: 41.2, + component_costs_usd: { uncached_input: 30, cache_read: 0, cache_write: 0, output: 11.2, reasoning: 0 }, + usage_complete: true, + usage_incomplete_reasons: [], + pricing_complete: false, + pricing_incomplete_reasons: [{ code: "model-pricing-unavailable" as const, model: "model-b" }], + partial_pricing: true, + cache_read_pricing_estimated: false, + event_count: 2, + priced_event_count: 1, + unpriced_event_count: 1, + models: ["model-a", "model-b"], + agents: ["agent-a"] + }; + const segment = { + ...summary, + control_generation: DIGEST, + workflow_run_id: "workflow-1", + source_event_sequences: [1, 2], + attempts: [ + { node_id: "review", iteration: 0, attempt: 0 }, + { node_id: "final-report", iteration: 0, attempt: 0 } + ] + }; + return { + ...metadata(), + workflow_ids: ["workflow-1"], + workflow: { + run_id: "workflow-1", + compiled_run_id: "compiled-1", + name: "workflow", + path: "workflow.tsx", + evidence_path: "evidence.json", + expanded_graph_path: "expanded-graph.json", + config_path: "config.json", + input_path: "input.json", + tasks_path: "tasks.json", + control_integrity_path: "control-integrity.json", + control_generation: DIGEST, + workflow_link_id: "123e4567-e89b-42d3-a456-426614174000", + execution_snapshot_path: "execution-snapshot.json", + task_node_ids: ["review", "final-report"] + }, + accounting: { + schema_version: "ultrafuzz.accounting.v4", + source: "usage-ledger", + workflow_run_id: "workflow-1", + current: structuredClone(segment), + segments: [structuredClone(segment)], + cumulative: { ...summary, source_run_ids: [], ...cumulative }, + checkpoint: { + schema_version: "ultrafuzz.accounting-checkpoint.v1", + ledger_event_count: 2, + last_source_event_sequence: 2, + control_generation: DIGEST, + workflow_run_id: "workflow-1" + }, + pricing_catalog: { + source: "configured-catalog", + status: "available", + fetched_at: CREATED_AT, + resolved_models: ["model-a"], + unresolved_models: ["model-b"], + model_prices: { "model-a": { inputUsdPerMillion: 1, outputUsdPerMillion: 2 } } + }, + updated_at: FINISHED_AT + } + }; +} + +function runSummaryLines(markdown: string): string[] { + return markdown + .split("\n") + .filter((line) => /^- (?:Elapsed time|Models used|Tokens used|Estimated spend):/u.test(line)); +} + +test("terminal projection restates whole-run accounting instead of the report-start snapshot", () => { + const input = { + completion: completion(), + state: { ...terminalState(), finished_at: "2026-09-01T06:02:00.000Z" }, + metadata: metadataWithAccounting(), + agentReport: agentReport() + }; + const result = projectTerminalReport(input); + assert.deepEqual(runSummaryLines(result.markdown), [ + "- Elapsed time: `6h 02m`", + "- Models used: `model-a, model-b`", + "- Tokens used: `12,345,678`", + "- Estimated spend: `$41.20+`" + ]); + assert.equal((result.report.run_metadata as Record).partial_pricing, true); + assert.deepEqual(result.report.issues, agentReport().issues); + + // Values the run records do not have keep the agent's copy. + const sparse = projectTerminalReport({ + ...input, + metadata: metadataWithAccounting({ estimated_spend: "unavailable", models: [] }) + }); + assert.deepEqual(runSummaryLines(sparse.markdown), [ + "- Elapsed time: `6h 02m`", + "- Models used: `example-model`", + "- Tokens used: `12,345,678`", + "- Estimated spend: `$0.01`" + ]); + assert.equal((sparse.report.run_metadata as Record).partial_pricing, false); +}); diff --git a/packages/runtime/test/unverified-report.test.ts b/packages/runtime/test/unverified-report.test.ts index 6e7456ee3..80fbb773d 100644 --- a/packages/runtime/test/unverified-report.test.ts +++ b/packages/runtime/test/unverified-report.test.ts @@ -231,3 +231,59 @@ test("failure to save the status cache does not suppress explicit report access" assert.equal(loadReportSnapshot(root).terminal, true); assert.equal(readReportPublicationStatus(root, state).status, "unknown"); }); + +test("unchecked reports restate whole-run accounting from run.json", () => { + const root = reportRun("whole-run-accounting"); + writeAgentReport(root); + const runId = path.basename(root); + const state = JSON.parse(fs.readFileSync(path.join(root, "state.json"), "utf8")); + fs.writeFileSync( + path.join(root, "state.json"), + JSON.stringify({ ...state, finished_at: "2026-09-01T03:30:00.000Z" }) + ); + fs.writeFileSync( + path.join(root, "run.json"), + JSON.stringify({ + run_id: runId, + created_at: "2026-09-01T00:00:00.000Z", + accounting: { + cumulative: { models: ["model-a"], tokens_used: "4,321", estimated_spend: "$3.00", partial_pricing: false } + } + }) + ); + const report = loadReportSnapshot(root); + assert.equal(report.verification, "not-checked"); + assert.match(report.markdown, /^- Elapsed time: `3h 30m`$/mu); + assert.match(report.markdown, /^- Models used: `model-a`$/mu); + assert.match(report.markdown, /^- Tokens used: `4,321`$/mu); + assert.match(report.markdown, /^- Estimated spend: `\$3\.00`$/mu); + assert.equal(reportSchema.parse(report.json).run_metadata.partial_pricing, false); + assertReportSnapshotRemainedCurrent(report); +}); + +test("unchecked reports render the run's goal-search census", () => { + const root = reportRun("goal-search-census"); + writeAgentReport(root); + const censusPath = path.join(root, "goal-search-coverage.json"); + fs.writeFileSync( + censusPath, + JSON.stringify({ + schema_version: "ultrafuzz.goal-search-coverage.v1", + run_id: path.basename(root), + totals: { planned: 2 }, + goals: [ + { node_id: "dynamic:class:1", logical_node_id: "class-goals", status: "completed-no-findings" }, + { node_id: "dynamic:class:2", logical_node_id: "class-goals", status: "stopped-early" } + ] + }) + ); + const report = loadReportSnapshot(root); + assert.match(report.markdown, /^- Targeted goal search lanes: `2`\n- Completed with a verified result: `1`$/mu); + assert.doesNotMatch(report.markdown, /Goal search coverage is unknown/u); + + // A census the reader refuses still leaves the report readable, with coverage stated as unknown. + const outside = path.join(temporaryRoot("ultrafuzz-outside-"), "goal-search-coverage.json"); + fs.renameSync(censusPath, outside); + fs.symlinkSync(outside, censusPath); + assert.match(loadReportSnapshot(root).markdown, /Goal search coverage is unknown/u); +}); diff --git a/packages/runtime/test/verified-output.test.ts b/packages/runtime/test/verified-output.test.ts index 89a23a7cc..8046cb947 100644 --- a/packages/runtime/test/verified-output.test.ts +++ b/packages/runtime/test/verified-output.test.ts @@ -245,7 +245,14 @@ test("terminal presentation discloses tolerated failures and preserves verified assert.equal(published.completion?.counts.failed, 1); assert.equal(published.completion?.outcome, "partial"); const expected = JSON.parse(fixture.reportBytes.toString("utf8")) as Record; - assert.deepEqual(published.json, { ...expected, completion: published.completion }); + // The run summary restates elapsed time from run.json and state.json; all review content is the agent's. + const elapsed = (published.json as { run_metadata: { elapsed_time: string } }).run_metadata.elapsed_time; + assert.match(elapsed, /^\d+\.\ds$/u); + assert.deepEqual(published.json, { + ...expected, + run_metadata: { ...(expected.run_metadata as Record), elapsed_time: elapsed }, + completion: published.completion + }); assert.match(published.markdown, /producer/u); }); From b6a2a6805ace3bda88677648de71f51badabcb29 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Mon, 28 Sep 2026 22:58:29 +0000 Subject: [PATCH 2/9] fix(runtime): stop the canonical report renderer rejecting its own output projectCanonicalFinalReport re-validates the Markdown it just rendered with regex rules written for hand-authored reports. publicProse did not escape `](`, so upstream prose that the review gates require byte-for-byte (a spec link, an image, `handlers[id](payload)`, a relative link whose `_` gets escaped) tripped the link or image rule. The link rule also rejected the renderer's own index anchors for any issue title with a non-ASCII letter, and the public projection threw when a redacted path was followed by `(` (`[redacted-path](line 12)`). The agent cannot repair byte-preserved fields, so every retry failed the same way and the run ended without a report. publicProse now escapes `(` after every `]`, so prose cannot form an inline link or image, and the image and link rules are deleted; the raw-HTML rule stays. Rendering of prose without `](` is unchanged, so existing reports keep verifying. The audit-context renderer (appendAuditContext, safeReportLink, SAFE_REPORT_RELATIVE_LINK_PATTERN) is deleted: report@3 has no audit_context field and rejects unknown keys, so it could never run. The prompt's `## Audit context` instructions, which asked the agent to write a section that the byte-exact canonical render never contains, are removed. Refs #1151 Co-Authored-By: Claude Opus 5.5 --- .ultrafuzz/prompts/review/final-report.md | 19 ---- packages/runtime/src/final-report-markdown.ts | 41 ++------- .../test/final-report-markdown.test.ts | 87 +++++++++++++++++++ 3 files changed, 93 insertions(+), 54 deletions(-) diff --git a/.ultrafuzz/prompts/review/final-report.md b/.ultrafuzz/prompts/review/final-report.md index 70008c7ae..ab9a4965c 100644 --- a/.ultrafuzz/prompts/review/final-report.md +++ b/.ultrafuzz/prompts/review/final-report.md @@ -385,25 +385,8 @@ Ultrafuzz is an automated smart-contract fuzzing campaign assistant. Issues belo - Tokens used: `` - Estimated spend: `` - Audit profile: `` - -## Audit context - -- Threat model: [THREAT_MODEL.md](); [threat-model.json]() -- Goal plan: [goal-plan.json]() ``` -Render `## Audit context` with exactly this heading, bullet order, and link -text, immediately after `## Run summary`. Use repository-relative or -report-relative paths to the run's own `threat-model` and `goal-plan` artifacts; -never absolute paths or external URLs. Omit an individual link whose artifact -the run did not produce, omit the `Goal plan` bullet when there is no goal plan, -and omit the whole section when the run produced none of them. Do not invent a -different heading, ordering, or link text: `ultrafuzz report` regenerates this -exact section deterministically from the run's own artifacts and overwrites -anything else. -Keep detailed threat content in those dedicated artifacts; do not duplicate it -in `report.md`. - Each production issue entry must use exactly this Markdown section order. The following example is structural only; replace the title, actor names, actions, outcomes, explanations, code, variants, and strategy IDs with issue-specific @@ -825,8 +808,6 @@ Before finishing, verify that: - `report.md` contains `## Property provenance`, including every property-derived finding and no invented property IDs for non-property findings. -- `report.md` renders the fixed `## Audit context` section for every artifact - the run produced, without copying their detailed analysis. - `report.md` contains `## Property implementation coverage` rendered from the exact runtime-authoritative coverage object. - `report.md` contains `## Goal search coverage` with counts recomputed from the diff --git a/packages/runtime/src/final-report-markdown.ts b/packages/runtime/src/final-report-markdown.ts index e60e04716..03c17f2b2 100644 --- a/packages/runtime/src/final-report-markdown.ts +++ b/packages/runtime/src/final-report-markdown.ts @@ -21,7 +21,6 @@ import { redactSecretsInText, type SecretScanMode } from "@ultrafuzz/security"; export const MAX_FINAL_REPORT_JSON_BYTES = 64 * 1024 * 1024; export const MAX_FINAL_REPORT_MARKDOWN_BYTES = 16 * 1024 * 1024; -const SAFE_REPORT_RELATIVE_LINK_PATTERN = /^\.\.\/(?:(?!\.\.?\/)[A-Za-z0-9._-]+\/)+(?!\.\.?$)[A-Za-z0-9._-]+$/u; /** * Secret placeholder for the public projection only. Two constraints pick it: * @@ -29,10 +28,9 @@ const SAFE_REPORT_RELATIVE_LINK_PATTERN = /^\.\.\/(?:(?!\.\.?\/)[A-Za-z0-9._-]+\ * the placeholder must be a fixed point of the redaction pass. The key-name assignment rule's * unquoted value class stops at whitespace, `,`, `;`, `]`, and `}`, so a placeholder containing * any of those is re-redacted on the next pass (`token=[redacted]` becomes `token=[redacted]]`). - * - The final-review Markdown gate rejects raw HTML, images, and links outside fenced code, and - * redacted values land unescaped in inline code (run summary values, coverage paths, source - * nodes), so the placeholder must not read as HTML (``), a link (`[redacted](`), - * an image, or emphasis (`*`, `_`). + * - The final-review Markdown gate rejects raw HTML outside fenced code, including inside inline + * code, where redacted values land unescaped (run summary values, coverage paths, source nodes), + * so the placeholder must not read as HTML (``). * * A bare uppercase word satisfies both. Every other redaction keeps the security package's default. */ @@ -362,12 +360,7 @@ function finalReportProseDirectiveViolation(prose: string): string | undefined { /(?:^|\n)- (?:Strategy loops|Audit profile catalog digest|Topology digest|Prompt digest|Expanded graph fingerprint):/imu, "contains legacy run metadata" ], - [/<[A-Za-z][^>]*>/u, "contains raw HTML outside fenced code"], - [/!\[[^\]]*\]\(/u, "contains an embedded image outside fenced code"], - [ - /(?]*>/u, "contains raw HTML outside fenced code"] ]; return forbiddenPatterns.find(([pattern]) => pattern.test(prose))?.[1]; } @@ -695,7 +688,6 @@ function renderCanonicalReport(report: JsonRecord, goalSearchCoverage: unknown): lines, isRecord(report.run_metadata) ? report.run_metadata.artifact_validation_warnings : undefined ); - appendAuditContext(lines, report.audit_context); const campaignDidNotRun = appendCampaignOutcome(lines, report.campaign_outcome); appendCoverageEvidence(lines, report.coverage_evidence); const goalCoverage = summarizeGoalSearchCoverage(goalSearchCoverage); @@ -961,29 +953,6 @@ function appendRunSummary(lines: string[], metadata: JsonRecord): void { } } -function appendAuditContext(lines: string[], value: unknown): void { - if (!isRecord(value)) return; - const threat = recordField(value, "threat_model"); - const goalPlan = recordField(value, "goal_plan"); - const threatMarkdown = safeReportLink(threat?.markdown); - const threatJson = safeReportLink(threat?.json); - const goalPlanJson = safeReportLink(goalPlan?.json); - if (threatMarkdown === undefined && threatJson === undefined && goalPlanJson === undefined) return; - lines.push("", "## Audit context", ""); - if (threatMarkdown !== undefined || threatJson !== undefined) { - const links = [ - threatMarkdown === undefined ? undefined : `[THREAT_MODEL.md](${threatMarkdown})`, - threatJson === undefined ? undefined : `[threat-model.json](${threatJson})` - ].filter((entry): entry is string => entry !== undefined); - lines.push(`- Threat model: ${links.join("; ")}`); - } - if (goalPlanJson !== undefined) lines.push(`- Goal plan: [goal-plan.json](${goalPlanJson})`); -} - -function safeReportLink(value: unknown): string | undefined { - return typeof value === "string" && SAFE_REPORT_RELATIVE_LINK_PATTERN.test(value) ? value : undefined; -} - function appendProductionIssue(lines: string[], rendered: RenderedIssue): void { const { issue } = rendered; lines.push("", renderedIssueHeading(rendered), "", publicProse(issueDescription(issue)), "", "### Severity", ""); @@ -1522,6 +1491,7 @@ function recordTitle(record: JsonRecord, fallback: string): string { return typeof record.title === "string" && record.title.trim().length > 0 ? record.title.trim() : fallback; } +/** Escaping `(` after every `]` keeps byte-preserved prose from forming an inline link or image. */ function publicProse(value: string): string { return value .replace(/\s+/gu, " ") @@ -1533,6 +1503,7 @@ function publicProse(value: string): string { .replaceAll("!", "\\!") .replaceAll("#", "\\#") .replaceAll("~", "\\~") + .replaceAll("](", "]\\(") .replaceAll("<", "<") .replaceAll(">", ">"); } diff --git a/packages/runtime/test/final-report-markdown.test.ts b/packages/runtime/test/final-report-markdown.test.ts index 9c1dd9a5e..ff8555706 100644 --- a/packages/runtime/test/final-report-markdown.test.ts +++ b/packages/runtime/test/final-report-markdown.test.ts @@ -27,6 +27,7 @@ test("final reports retain artifact warnings and their context without changing import { validateSafeId, type ReportCompletion } from "@ultrafuzz/artifacts"; import { redactSecretsInText } from "@ultrafuzz/security"; +import { fromMarkdown } from "mdast-util-from-markdown"; import { isDirectiveConformingFinalReportMarkdown, @@ -1406,3 +1407,89 @@ test("goal search coverage is rendered into the Markdown report instead of only ); assert.match(roamingOnly.markdown, /^No issues were reported, but no targeted goal search lane ran, /mu); }); + +interface MarkdownNode { + type: string; + url?: string; + value?: string; + children?: MarkdownNode[]; +} + +/** Flatten the CommonMark tree so assertions describe what a reader sees, not escape bytes. */ +function markdownNodes(markdown: string): Array<{ type: string; url?: string; text: string }> { + const text = (node: MarkdownNode): string => node.value ?? (node.children ?? []).map(text).join(""); + const nodes: Array<{ type: string; url?: string; text: string }> = []; + const walk = (node: MarkdownNode): void => { + nodes.push({ type: node.type, ...(node.url === undefined ? {} : { url: node.url }), text: text(node) }); + for (const child of node.children ?? []) walk(child); + }; + walk(fromMarkdown(markdown) as unknown as MarkdownNode); + return nodes; +} + +test("upstream prose with link or image syntax renders as literal text", () => { + const report = renderableReport(); + const [issue] = report.issues as Array>; + if (issue === undefined) throw new Error("missing issue fixture"); + const description = + "See [the spec](https://example.com/spec), ![flow](https://example.com/flow.png), " + + "[THREAT_MODEL.md](../threat-model/THREAT_MODEL.md), and handlers[id](payload)."; + issue.description = description; + (issue.proof_of_concept as Record).scenario = ["Call handlers[id](payload).", "Observe it."]; + const before = structuredClone(report); + + const projection = projectCanonicalFinalReport(report); + assert.deepEqual(projection.report, before); + const nodes = markdownNodes(projection.markdown); + assert.deepEqual( + nodes.filter((node) => node.type === "image" || (node.type === "link" && !node.url?.startsWith("#"))), + [] + ); + assert.ok(nodes.some((node) => node.type === "paragraph" && node.text === description)); + assert.ok(nodes.some((node) => node.type === "paragraph" && node.text === "Call handlers[id](payload).")); +}); + +test("issue titles with non-ASCII letters render with index anchors that resolve to their headings", () => { + // GitHub heading slugs: lowercase, keep letters, marks, digits, spaces, "-" and "_", then spaces become "-". + const slug = (text: string): string => + text + .trim() + .toLowerCase() + .replace(/[^\p{L}\p{M}\p{N}\s_-]/gu, "") + .replace(/\s/gu, "-"); + for (const title of ["Δ-neutral rebalance drifts", "Naïve [share] math"]) { + const report = renderableReport(); + const [issue] = report.issues as Array>; + if (issue === undefined) throw new Error("missing issue fixture"); + issue.title = `[L-01] - ${title}`; + for (const entry of report.property_provenance as Array>) entry.title = issue.title; + + const nodes = markdownNodes(projectCanonicalFinalReport(report).markdown); + const headings = new Set(nodes.filter((node) => node.type === "heading").map((node) => slug(node.text))); + const anchors = nodes.filter((node) => node.type === "link" && node.url?.startsWith("#")); + assert.ok(anchors.length > 0, title); + for (const anchor of anchors) assert.ok(headings.has(anchor.url?.slice(1) ?? ""), `${title}: ${anchor.url ?? ""}`); + } +}); + +test("public projection keeps a redacted path followed by a parenthesis as literal text", () => { + const report = renderableReport(); + const [issue] = report.issues as Array>; + if (issue === undefined) throw new Error("missing issue fixture"); + issue.description = "The reproducer at /srv/customer/private/Repro.t.sol(line 12) fails."; + + const published = projectPublicCanonicalFinalReport(report); + assert.equal( + (published.report.issues as Array>)[0]?.description, + "The reproducer at [redacted-path](line 12) fails." + ); + const nodes = markdownNodes(published.markdown); + assert.equal( + nodes.some((node) => node.type === "link" && !node.url?.startsWith("#")), + false + ); + assert.ok( + nodes.some((node) => node.type === "paragraph" && node.text === "The reproducer at [redacted-path](line 12) fails.") + ); + assertPublicProjectionFixedPoint(published); +}); From c51b310a4a9cd3c473c5fa89504ebe0a8be6fb59 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Mon, 28 Sep 2026 22:58:29 +0000 Subject: [PATCH 3/9] docs: describe whole-run report summaries and literal prose links The artifacts reference claimed report.md links to THREAT_MODEL.md, threat-model.json and goal-plan.json (only dead code ever did) and that terminal presentations keep the report-start accounting snapshot. Document what runtime presentations now restate, that prose link syntax renders as literal text, and the upgrade effect on existing terminal publications. Refs #1151 Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 1 + docs/reference/artifacts-reports.md | 35 +++++++++++++++++------------ 2 files changed, 22 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 960ba9394..732c71a9d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,6 +12,7 @@ ### Other changes +- **[runtime] [prompts] [docs]** Runtime report presentations (the verified terminal publication and unchecked reports) now restate elapsed time, models, tokens, and estimated spend from `run.json` and the recorded finish time instead of the snapshot the report agent received when its task started, and unchecked reports render the run's goal-search census instead of always stating unknown coverage. The canonical renderer escapes `](` in prose, so it no longer rejects its own output when byte-preserved upstream prose contains link or image syntax, when an issue title contains a non-ASCII letter, or when the public projection redacts a path followed by `(`; the dead audit-context renderer and the prompt's contradictory `## Audit context` instructions are removed. Markdown for reports without `](` in prose is byte-identical, but terminal publications written by earlier versions read back as `not-checked` once the restated summary differs from their stored bytes (usually at least the elapsed time), and an agent report whose prose held a previously accepted `](` form (a `#anchor` or `../` link, or `\](`) no longer re-verifies (#1151). - **[runtime]** `resume --refresh-controller` on a dynamically-expanded run now publishes a task manifest that still re-derives from its sealed base, closing a data-integrity regression. Current-controller rendering keeps binding each plan-time prompt to its authenticated retained snapshot (#980), but that execution-only binding now reaches the task-spec literal alone instead of the compiled task manifest, whose `renderedPromptPath` is normalized back to the sealed launch path recorded in `plan.json`. Previously the refreshed controller published the rebound `prompt-snapshots/.md` path into `smithers/tasks.json`, while `verifyDynamicRuntimeMaterialization` re-derived the same document from the sealed `controls/runtime-base-tasks.json` carrying the launch path; the two fingerprinted differently and every gated command (`sync`, `pause`, replay, fork) failed `WORKFLOW_CONTROL_EVIDENCE_INVALID` for the rest of the run's life, with no self-heal on a repeat refresh. Because resume is ungated, a run already damaged this way is recoverable: the normalization scrubs a rebound manifest on the next refresh, including when the sealed base manifest is absent and the live document is the compile input. Prompt authentication is unchanged -- a retained snapshot whose bytes no longer match `plan.json`'s recorded digest still aborts the refresh, and a task with no plan row still fails closed rather than falling back to the cleanup-owned launch path (#1002). - Pi model profiles now accept and forward the CLI's `max` thinking level, in addition to the previously supported levels through `xhigh`. - Upgrades the pinned workflow engine to Smithers 0.35.0. The Effect 4 tree is unchanged at `4.0.0-beta.105`, so the `@effect/*` override family and the `pnpm-workspace.yaml` mirror keep their exact pins; the deterministic npm resolution cutoff moves to `2026-08-18T06:00:00Z`, past `smthrs@0.35.0`'s publish instant. Six compatibility patches are re-derived against upstream's new `smithersRuntimeSpawn`/`smithersRuntimeReentry` indirection and the re-nested `adapter.insertRun` call; none is retired, because every workaround they encode is still absent upstream. Ultrafuzz's Smithers state mirrors gain the new `succeeded-with-failures` run state and `stalled` node state, and the inspect contract admits `tokenUsage`, `run.cancellationSource` and `runState.warnings`. Existing `0.34.0` dependency manifests migrate forward in place at `init`, so an already-scaffolded project converges onto the new pin without `--force` regenerating the run. Native resume never read the target project's `.smithers/package.json` -- launch admission asserts the manifest only against the operator controller's own freshly rendered temporary root -- so a stopped run already resumed under its own run ID and continues to. Ultrafuzz's strict readers for `why`, `status`, `node` and the `TokenUsageReported` event stream also admit the release's new `warnings`, `counts.stalled`, `tokenUsage.freshInputTokens`, `freshInputTokens`/`costUsd` and `stalled` blocker fields, and its two new `NodeStalled`/`RunConcurrencySaturated` event types; a stalled node now resets under `resume --retry-failed` alongside failed ones, and the evmbench adapter no longer aborts a converged run on the release's widened `degraded` verdict. The runner adds four forward-only SQLite migrations (0041-0044) (#955). diff --git a/docs/reference/artifacts-reports.md b/docs/reference/artifacts-reports.md index 2683bb70e..dff578046 100644 --- a/docs/reference/artifacts-reports.md +++ b/docs/reference/artifacts-reports.md @@ -537,8 +537,9 @@ artifacts/final-report/report.json `report.json` must satisfy `ultrafuzz/report@3` with the exact `ultrafuzz.report.v3` version literal. After a run stops, the runtime can format -that report and attach whole-run completion information without changing the -agent's files. Verified publications use: +that report, attach whole-run completion information, and restate the run +summary's elapsed time and accounting without changing the agent's files. +Verified publications use: ```text review/runtime-report//report.json @@ -741,11 +742,12 @@ this section with the typed handoff and rejects missing, duplicated, reordered, or bare coverage scores. Raw `covg-eval` output is for iteration only and defines neither published declaration-completeness view. -Current-run `report.md` contains concise links to `THREAT_MODEL.md`, -`threat-model.json`, and `goal-plan.json`, plus source-node provenance for each -production issue. Detailed threat analysis stays in the dedicated threat-model -artifacts and is not duplicated into the report. `report.json` preserves the -same `source_nodes` arrays. +Current-run `report.md` contains source-node provenance for each production +issue and does not link to other run files. Detailed threat analysis stays in +the dedicated threat-model artifacts and is not duplicated into the report. +`report.json` preserves the same `source_nodes` arrays. Inline link and image +syntax inside report prose, including prose preserved byte-for-byte from +upstream findings, renders as literal text. When workflow usage data is available, run metadata includes `accounting.cumulative.tokens_used` and @@ -755,13 +757,18 @@ available cumulative values into the markdown run summary and into persisted estimate is partial because some token usage did not have pricing data. -If cumulative metadata has not synchronized when the final-report producer -starts, its live Smithers fallback is a snapshot through that producer's start. -It includes earlier attempts but cannot include the producer's own eventual -duration, model fallback, tokens, or cost. A terminal presentation of an existing -verified agent report preserves those accounting values. Report v3 has no -metric-scope field, so use -`ultrafuzz stats` after terminal synchronization for closed-run accounting. +The final-report producer receives its run summary when its task starts: from +cumulative metadata when it has synchronized, otherwise from a live Smithers +fallback. Either way it is a snapshot through that producer's start. It includes +earlier attempts but cannot include the producer's own eventual duration, model +fallback, tokens, or cost, and the agent's `report.json` and `report.md` keep +that snapshot. Runtime presentations (the verified terminal publication and +unchecked reports) restate the run summary instead: elapsed time from +`run.json#created_at` to `state.json#finished_at`, and models, tokens, +estimated spend, and `partial_pricing` from the current +`accounting.cumulative`. A value those records lack, or record as +`unavailable`, keeps the agent's copy. Use `ultrafuzz stats` for the full +accounting breakdown. `accounting.segments` publishes one rollup per checkpoint generation, and `accounting.current` identifies the latest segment. Each segment retains every From 628f4c436922b4601a62f57dea7d46f6711ec50e Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Mon, 28 Sep 2026 23:10:39 +0000 Subject: [PATCH 4/9] test(runtime): pin why prose escapes the parenthesis rather than the bracket Escaping `](` as `\](` would break the public projection on two shapes: after publicProse doubles a prose backslash, `\\\](` matches the UNC private-path pattern, and `token=REDACTED\](` extends the redacted assignment value so the fixed-point re-scan redacts it again. Both public projections succeed with the `]\(` escape; with `\](` this test fails. Refs #1151 Co-Authored-By: Claude Opus 5.5 --- packages/runtime/test/final-report-markdown.test.ts | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/packages/runtime/test/final-report-markdown.test.ts b/packages/runtime/test/final-report-markdown.test.ts index ff8555706..215bc4cc7 100644 --- a/packages/runtime/test/final-report-markdown.test.ts +++ b/packages/runtime/test/final-report-markdown.test.ts @@ -1492,4 +1492,11 @@ test("public projection keeps a redacted path followed by a parenthesis as liter nodes.some((node) => node.type === "paragraph" && node.text === "The reproducer at [redacted-path](line 12) fails.") ); assertPublicProjectionFixedPoint(published); + + // An escape inserted before `]` would read as a UNC path after a doubled backslash, and would + // extend a redacted assignment value so the fixed-point re-scan redacts it again. + for (const description of ["Match a literal \\](x) in the parser.", "Set token=synthetic-escape-secret](x)."]) { + issue.description = description; + assertPublicProjectionFixedPoint(projectPublicCanonicalFinalReport(report)); + } }); From b964ae2301582ecfdc5fa0484e2f80fd09d11dbd Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 02:46:47 +0000 Subject: [PATCH 5/9] fix(runtime): restate run-summary tokens, spend and partial pricing together withWholeRunSummary restated tokens_used from run.json's cumulative accounting but kept the agent's estimated_spend and partial_pricing whenever the cumulative spend was "unavailable". For a direct run, that agent spend can come from the live Smithers fallback taken when the report task started, which may price only part of the run. The runtime presentation then showed whole-run tokens beside a report-start spend, with no partial marker. Tokens, spend and partial pricing now move as one group: when the cumulative record carries a token count, all three come from it, and a whole-run spend recorded as "unavailable" stays "unavailable". Without cumulative accounting, all three keep the agent's copy as before. The sparse test case pinned the mixed summary, using a cumulative block (estimated_spend "unavailable" beside estimated_spend_usd 41.2) that the runtime's accounting validator rejects. It now uses a ledger that priced no event, and the fixture's pricing markers carry the component field the validator requires. The fixture passes storedAccountingDocument (via assertRunMetadataAccountingUsageAuthority, which stops only at the ledger comparison). Co-Authored-By: Claude Opus 5.5 --- docs/reference/artifacts-reports.md | 8 +-- .../runtime/src/terminal-report-projection.ts | 10 ++-- .../test/terminal-report-projection.test.ts | 53 +++++++++++-------- 3 files changed, 43 insertions(+), 28 deletions(-) diff --git a/docs/reference/artifacts-reports.md b/docs/reference/artifacts-reports.md index dff578046..7235995eb 100644 --- a/docs/reference/artifacts-reports.md +++ b/docs/reference/artifacts-reports.md @@ -766,9 +766,11 @@ that snapshot. Runtime presentations (the verified terminal publication and unchecked reports) restate the run summary instead: elapsed time from `run.json#created_at` to `state.json#finished_at`, and models, tokens, estimated spend, and `partial_pricing` from the current -`accounting.cumulative`. A value those records lack, or record as -`unavailable`, keeps the agent's copy. Use `ultrafuzz stats` for the full -accounting breakdown. +`accounting.cumulative`. Tokens, estimated spend, and `partial_pricing` are +restated together whenever `accounting.cumulative` records a token count, so a +whole-run spend recorded as `unavailable` stays `unavailable` instead of showing +the agent's report-start figure. Otherwise, a value those records lack keeps the +agent's copy. Use `ultrafuzz stats` for the full accounting breakdown. `accounting.segments` publishes one rollup per checkpoint generation, and `accounting.current` identifies the latest segment. Each segment retains every diff --git a/packages/runtime/src/terminal-report-projection.ts b/packages/runtime/src/terminal-report-projection.ts index ebc6094cd..a32da8800 100644 --- a/packages/runtime/src/terminal-report-projection.ts +++ b/packages/runtime/src/terminal-report-projection.ts @@ -61,8 +61,10 @@ export function projectTerminalReport(input: TerminalReportProjectionInput): Can /** * The report agent copies a run summary that the host captured when the report task started, so its * elapsed time and accounting miss the report task itself and anything that finished later. Runtime - * presentations restate them from run.json and the recorded finish time. A value those records do - * not provide keeps the agent's copy; malformed records are ignored, never thrown. + * presentations restate them from run.json and the recorded finish time. Tokens, spend, and partial + * pricing move together, because the agent's spend may price only part of the run; a spend the + * whole-run record calls unavailable stays unavailable. A value those records do not provide keeps + * the agent's copy; malformed records are ignored, never thrown. */ export function withWholeRunSummary( runMetadata: Record, @@ -76,9 +78,9 @@ export function withWholeRunSummary( const models = field(cumulative, "models"); if (isNonEmptyStringList(models)) summary.models_used = [...models]; const tokens = field(cumulative, "tokens_used"); - if (availableLabel(tokens)) summary.tokens_used = tokens; const spend = field(cumulative, "estimated_spend"); - if (availableLabel(spend)) { + if (availableLabel(tokens) && typeof spend === "string" && spend.trim() !== "") { + summary.tokens_used = tokens; summary.estimated_spend = spend; summary.partial_pricing = field(cumulative, "partial_pricing") === true; } diff --git a/packages/runtime/test/terminal-report-projection.test.ts b/packages/runtime/test/terminal-report-projection.test.ts index 43c7e5b3e..018cafce4 100644 --- a/packages/runtime/test/terminal-report-projection.test.ts +++ b/packages/runtime/test/terminal-report-projection.test.ts @@ -197,8 +197,12 @@ test("terminal projection enforces bounded and internally consistent completion assert.match(result.markdown, /identities omitted from this bounded census: `1`/u); }); -/** A valid run.json whose cumulative accounting already includes the report task's own usage. */ -function metadataWithAccounting(cumulative: { estimated_spend?: string; models?: string[] } = {}): RunMetadataDocument { +/** + * A valid run.json whose cumulative accounting already includes the report task's own usage. When + * `priced` is false the usage ledger priced no event, so the whole-run spend is unavailable. + */ +function metadataWithAccounting(priced = true): RunMetadataDocument { + const unpricedModels = priced ? ["model-b"] : ["model-a", "model-b"]; const summary = { uncached_input_tokens: 9_000_000, input_tokens: 9_000_000, @@ -210,18 +214,27 @@ function metadataWithAccounting(cumulative: { estimated_spend?: string; models?: billable_token_total: 12_345_678, total_tokens: 12_345_678, tokens_used: "12,345,678", - estimated_spend: "$41.20+", - estimated_spend_usd: 41.2, - component_costs_usd: { uncached_input: 30, cache_read: 0, cache_write: 0, output: 11.2, reasoning: 0 }, + ...(priced + ? { + estimated_spend: "$41.20+", + estimated_spend_usd: 41.2, + component_costs_usd: { uncached_input: 30, cache_read: 0, cache_write: 0, output: 11.2, reasoning: 0 } + } + : { + estimated_spend: "unavailable", + component_costs_usd: { uncached_input: 0, cache_read: 0, cache_write: 0, output: 0, reasoning: 0 } + }), usage_complete: true, usage_incomplete_reasons: [], pricing_complete: false, - pricing_incomplete_reasons: [{ code: "model-pricing-unavailable" as const, model: "model-b" }], + pricing_incomplete_reasons: (["output", "uncached_input"] as const).flatMap((component) => + unpricedModels.map((model) => ({ code: "model-pricing-unavailable" as const, component, model })) + ), partial_pricing: true, cache_read_pricing_estimated: false, event_count: 2, - priced_event_count: 1, - unpriced_event_count: 1, + priced_event_count: priced ? 1 : 0, + unpriced_event_count: priced ? 1 : 2, models: ["model-a", "model-b"], agents: ["agent-a"] }; @@ -260,7 +273,7 @@ function metadataWithAccounting(cumulative: { estimated_spend?: string; models?: workflow_run_id: "workflow-1", current: structuredClone(segment), segments: [structuredClone(segment)], - cumulative: { ...summary, source_run_ids: [], ...cumulative }, + cumulative: { ...summary, source_run_ids: [] }, checkpoint: { schema_version: "ultrafuzz.accounting-checkpoint.v1", ledger_event_count: 2, @@ -272,9 +285,9 @@ function metadataWithAccounting(cumulative: { estimated_spend?: string; models?: source: "configured-catalog", status: "available", fetched_at: CREATED_AT, - resolved_models: ["model-a"], - unresolved_models: ["model-b"], - model_prices: { "model-a": { inputUsdPerMillion: 1, outputUsdPerMillion: 2 } } + resolved_models: priced ? ["model-a"] : [], + unresolved_models: unpricedModels, + model_prices: priced ? { "model-a": { inputUsdPerMillion: 1, outputUsdPerMillion: 2 } } : {} }, updated_at: FINISHED_AT } @@ -304,16 +317,14 @@ test("terminal projection restates whole-run accounting instead of the report-st assert.equal((result.report.run_metadata as Record).partial_pricing, true); assert.deepEqual(result.report.issues, agentReport().issues); - // Values the run records do not have keep the agent's copy. - const sparse = projectTerminalReport({ - ...input, - metadata: metadataWithAccounting({ estimated_spend: "unavailable", models: [] }) - }); - assert.deepEqual(runSummaryLines(sparse.markdown), [ + // A ledger that priced nothing makes the whole-run spend unavailable. The agent's report-start + // spend covers only part of the run, so it is not shown next to whole-run tokens. + const unpriced = projectTerminalReport({ ...input, metadata: metadataWithAccounting(false) }); + assert.deepEqual(runSummaryLines(unpriced.markdown), [ "- Elapsed time: `6h 02m`", - "- Models used: `example-model`", + "- Models used: `model-a, model-b`", "- Tokens used: `12,345,678`", - "- Estimated spend: `$0.01`" + "- Estimated spend: `unavailable`" ]); - assert.equal((sparse.report.run_metadata as Record).partial_pricing, false); + assert.equal((unpriced.report.run_metadata as Record).partial_pricing, true); }); From dde848fb9b1abfe5368a22f5aae57e831930bd31 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 02:47:00 +0000 Subject: [PATCH 6/9] fix(runtime): drop the renderer's legacy prose rules and escape `](` in warning codes The canonical renderer re-checks its own output. Its remaining prose rules were written for Markdown that agents used to write by hand, and they still matched byte-preserved upstream content. Each of these made projectCanonicalFinalReport throw, on this branch before this commit and on main: - a family variant titled "Item 1" (legacy numbered-item field); - a family variant titled "Source Node Id" (legacy source identifier); - a blocker summary "Critical" (unsupported Critical severity); - a blocker summary "Strategy loops: 4" (legacy run metadata). The final-report gate requires those fields to equal upstream, so the agent cannot repair them and every retry fails the same way. The other three rules (#### Sources, legacy report sections, legacy issue subsections) could not match rendered output at all, because publicProse escapes `#`. The self-check now keeps only its raw-HTML rule. Removing checks does not change any rendered bytes. artifact_validation_warnings[].code is free-form (1-128 characters) and is the one free-form field rendered as plain text without publicProse. With the link and image rules gone, a code such as `[notice](https://example.com/x)` rendered as a live link, in the report and in the public warnings companion. Only `](` is escaped there, so real gate codes such as ARTIFACT_OPTIONAL_METADATA_MISSING keep their bytes (full publicProse would escape their underscores). The three hand-mutated Markdown tests that used the deleted rules as probes now probe the raw-HTML rule through the same fence handling. Co-Authored-By: Claude Opus 5.5 --- packages/runtime/src/final-report-markdown.ts | 30 ++--- .../test/final-report-markdown.test.ts | 109 ++++++++++++------ 2 files changed, 83 insertions(+), 56 deletions(-) diff --git a/packages/runtime/src/final-report-markdown.ts b/packages/runtime/src/final-report-markdown.ts index 03c17f2b2..3a8909aed 100644 --- a/packages/runtime/src/final-report-markdown.ts +++ b/packages/runtime/src/final-report-markdown.ts @@ -258,9 +258,11 @@ function finalReportMarkdownDirectiveViolation(markdown: string, report: JsonRec if (!markdown.includes("\n## Property provenance\n")) { return "missing property provenance"; } + // Report prose is preserved byte-for-byte from upstream artifacts that the agent cannot repair, so + // the only rule left is one escaped prose cannot match: publicProse escapes `<`, and only + // unescaped inline-code values can still carry raw HTML. const prose = markdownOutsideFencedCode(markdown).replace(//giu, ""); - const proseViolation = finalReportProseDirectiveViolation(prose); - if (proseViolation !== undefined) return proseViolation; + if (/<[A-Za-z][^>]*>/u.test(prose)) return "contains raw HTML outside fenced code"; const rendered = renderedIssues(Array.isArray(report.issues) ? report.issues.filter(isRecord) : []); const expectedHeadings = rendered.map(renderedIssueHeading); const headings = markdown.split("\n").filter((line) => line.startsWith("## [")); @@ -345,26 +347,6 @@ function completionFindingsViolation( return undefined; } -function finalReportProseDirectiveViolation(prose: string): string | undefined { - // Critical is not a supported report severity, but the word remains valid in explanatory prose - // (for example, "a critical invariant"). Reject only a standalone severity-like label rather than - // rewriting or discarding the validated finding text. - const forbiddenPatterns: ReadonlyArray = [ - [/(?:^|\n)(?:#{1,6}\s+|-\s+)?(?:\*\*)?Critical(?:\*\*)?\s*$/imu, "contains the unsupported Critical severity"], - [/(?:^|\n)#### Sources\s*$/imu, "contains a legacy Sources section"], - [/\*\*Source (?:Node|Property) Id\*\*/iu, "contains a legacy source identifier field"], - [/(?:^|\n)- \*\*Item \d+\*\*/imu, "contains a legacy numbered-item field"], - [/(?:^|\n)## (?:Executive summary|Issue index|Additional report data)\s*$/imu, "contains a legacy report section"], - [/(?:^|\n)#{3,6} (?:Lifecycle|Strategy|Strategy provenance)\s*$/imu, "contains a legacy issue subsection"], - [ - /(?:^|\n)- (?:Strategy loops|Audit profile catalog digest|Topology digest|Prompt digest|Expanded graph fingerprint):/imu, - "contains legacy run metadata" - ], - [/<[A-Za-z][^>]*>/u, "contains raw HTML outside fenced code"] - ]; - return forbiddenPatterns.find(([pattern]) => pattern.test(prose))?.[1]; -} - function validateReport(report: unknown): JsonRecord { const serialized = `${JSON.stringify(report)}\n`; const validation = validateArtifactContract("ultrafuzz/report@3", serialized, "report.json"); @@ -830,8 +812,10 @@ function appendArtifactValidationWarnings(lines: string[], value: unknown): void "" ); for (const warning of value.filter(isRecord)) { + // Codes render as plain text. Only `](` is escaped, so the bytes of real gate codes do not change. + const code = inlineValue(warning.code).replaceAll("](", "]\\("); lines.push( - `- ${inlineValue(warning.code)} — \`${inlineValue(warning.artifact_path)}#${inlineValue(warning.field_path)}\`: ${publicProse(String(warning.message))}` + `- ${code} — \`${inlineValue(warning.artifact_path)}#${inlineValue(warning.field_path)}\`: ${publicProse(String(warning.message))}` ); if (warning.source_path !== undefined) lines.push(` - Available context: \`${inlineValue(warning.source_path)}\``); } diff --git a/packages/runtime/test/final-report-markdown.test.ts b/packages/runtime/test/final-report-markdown.test.ts index 215bc4cc7..fba566021 100644 --- a/packages/runtime/test/final-report-markdown.test.ts +++ b/packages/runtime/test/final-report-markdown.test.ts @@ -1082,7 +1082,7 @@ test("directive validation rejects injected or presentation-divergent Markdown", const projection = projectCanonicalFinalReport(renderableReport()); assert.equal( isDirectiveConformingFinalReportMarkdown( - `${projection.markdown}\n## Executive summary\n\nInjected presentation.\n`, + `${projection.markdown}\n\n`, projection.report ), false @@ -1094,17 +1094,13 @@ test("directive validation rejects injected or presentation-divergent Markdown", ), false ); - for (const obsoleteLine of [ - "- Strategy loops: `3`", - `- Prompt digest: \`${"a".repeat(64)}\``, - "### Strategy\n\n| Strategy | Detection rate |\n| --- | --- |\n| stateful-invariant | 2/3 |" - ]) { - assert.equal( - isDirectiveConformingFinalReportMarkdown(`${projection.markdown}\n${obsoleteLine}\n`, projection.report), - false, - obsoleteLine - ); - } + assert.equal( + isDirectiveConformingFinalReportMarkdown( + projection.markdown.replace("\n## Property provenance\n", "\n## Provenance\n"), + projection.report + ), + false + ); }); test("directive conformance requires both coverage headings without an opt-out", () => { @@ -1133,35 +1129,27 @@ test("directive validation treats fenced proof code as code while retaining pros const input = renderableReport(); const issue = (input.issues as Array>)[0]!; issue.proof_of_concept = { - scenario: ["Prepare the bounded state.", "Execute the transition and observe the mismatch."], + scenario: ["Prepare the state where balance exceeds .", "Execute the transition and observe the mismatch."], language: "solidity", code: [ - "contract CriticalStateProbe {", - ' string internal constant label = "#### Sources";', - " // **Source Node Id** and ### Strategy provenance are target identifiers here.", + "contract MarkupProbe {", + ' string internal constant label = "bold";', + " // is a target string here.", "}" ].join("\n") }; const projection = projectCanonicalFinalReport(input); - assert.match(projection.markdown, /contract CriticalStateProbe/u); - assert.match(projection.markdown, /#### Sources/u); + assert.match(projection.markdown, /", "~~~~ " @@ -1190,7 +1177,7 @@ test("directive validation recognizes CommonMark tilde fences and matching close isDirectiveConformingFinalReportMarkdown( tildeProof.replace( "~~~~ \n\n## Property implementation coverage", - "~~~~ \n\nCritical\n\n## Property implementation coverage" + "~~~~ \n\nbold\n\n## Property implementation coverage" ), projection.report ), @@ -1198,7 +1185,7 @@ test("directive validation recognizes CommonMark tilde fences and matching close ); assert.equal( isDirectiveConformingFinalReportMarkdown( - insertProofBlock([" ~~~solidity", "Critical", " ~~~"].join("\n")), + insertProofBlock([" ~~~solidity", "", " ~~~"].join("\n")), projection.report ), false, @@ -1449,6 +1436,62 @@ test("upstream prose with link or image syntax renders as literal text", () => { assert.ok(nodes.some((node) => node.type === "paragraph" && node.text === "Call handlers[id](payload).")); }); +test("artifact validation warning codes with link or image syntax render as literal text", () => { + const report = renderableReport(); + const codes = ["[notice](https://example.com/x)", "![t](https://example.com/p.png)"]; + const warnings = codes.map((code) => ({ + code, + artifact_path: "artifacts/dedupe/strategy-detections.json", + field_path: "$[5].family_id", + message: "Optional metadata is missing", + gate: "strategy-detection-review-stage-reconciliation" + })); + (report.run_metadata as Record).artifact_validation_warnings = warnings; + + for (const markdown of [ + projectCanonicalFinalReport(report).markdown, + projectPublicArtifactValidationWarnings(warnings).markdown + ]) { + const nodes = markdownNodes(markdown); + assert.deepEqual( + nodes.filter((node) => node.type === "image" || (node.type === "link" && !node.url?.startsWith("#"))), + [] + ); + for (const code of codes) { + assert.ok( + nodes.some((node) => node.type === "paragraph" && node.text.startsWith(`${code} — `)), + code + ); + } + } +}); + +test("upstream prose that reads like a legacy report label still renders", () => { + const report = renderableReport(); + const [issue] = report.issues as Array>; + if (issue === undefined) throw new Error("missing issue fixture"); + issue.family_variants = [ + { id: "variant-1", title: "Item 1", summary: "The first sibling path.", dedupe_key: "variant-1" }, + { id: "variant-2", title: "Source Node Id", summary: "Mislabelled identifiers.", dedupe_key: "variant-2" } + ]; + (report.property_implementation_coverage as Record).blocker_summaries = [ + "Critical", + "Strategy loops: 4" + ]; + + const items = markdownNodes(projectCanonicalFinalReport(report).markdown) + .filter((node) => node.type === "listItem") + .map((node) => node.text); + for (const text of [ + "Item 1: The first sibling path.", + "Source Node Id: Mislabelled identifiers.", + "Critical", + "Strategy loops: 4" + ]) { + assert.ok(items.includes(text), text); + } +}); + test("issue titles with non-ASCII letters render with index anchors that resolve to their headings", () => { // GitHub heading slugs: lowercase, keep letters, marks, digits, spaces, "-" and "_", then spaces become "-". const slug = (text: string): string => From 561a8815e29d3dbd151a271d314aa18f953200eb Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 02:47:00 +0000 Subject: [PATCH 7/9] docs(changelog): file the terminal-publication relabel under breaking changes Terminal report publications written before this release no longer verify, because readers re-render them and the restated run summary differs from the stored bytes. `ultrafuzz report` and the dashboard then serve a complete run as the unchecked PARTIAL presentation. `report bundle` falls back to a report-only archive. `--require-verified` reads and the Modal public worker fail. Sync does not re-publish, because a publication status already exists for that terminal state. This is a user-visible break, so it moves to "Breaking changes". The old entry also named `](#anchor)` among the prose forms main accepted. main rejected it, because publicProse escapes `#`. The forms main accepted were `\](` and `](` before an escape-free `../` path. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 732c71a9d..e1e34b60b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,7 @@ ### Breaking changes +- **[runtime] [cli] [dashboard] [modal]** Terminal report publications written before this release no longer verify, because readers re-render them and the restated run summary (at least its elapsed time) differs from the stored bytes. `ultrafuzz report` and the dashboard then show a complete run as `# Ultrafuzz report — PARTIAL` with verification `not-checked`, `report bundle` falls back to a report-only archive, and `--require-verified` reads and the Modal public worker fail for that run; nothing re-publishes it automatically. Agent reports whose prose contains `\](` or `](` before an escape-free `../` relative path also render differently and no longer re-verify (#1151). - **[runtime] [cli] [docs]** Restores native Smithers continuation as the normal `resume` behavior. Ordinary resume now sends the persisted workflow and same Smithers run ID directly to Smithers with workflow-change acceptance instead of requiring Ultrafuzz control seals, link journals, controller generations, current schema bindings, graph identity, or metadata projections. `--refresh-controller` renders the current controller beside the historical source and continues that same Smithers run; it no longer publishes an authenticated historical generation. Completed Smithers rows and artifact bytes are not reset, replayed, migrated, or rewritten, and missing optional final-report metadata renders as `unavailable`. Replay determinism after accepting changed workflow source is therefore Smithers' caller-visible responsibility. The controller re-finalization CLI option is removed; use native resume or the existing explicit reset/retry operations (#939). - **[config] [docs] [evals]** Removes the `full` audit profile and renames its packaged topology to `topologies/exhaustive.yml` (byte-identical, topology digest unchanged); the `exhaustive` profile now runs that complete specialist workflow at its existing maximum settings, and no longer follows an edited `.ultrafuzz/topology.yml` (a project `topology_path` or `--topology-path` still overrides it). Configurations that select `full` now fail with the standard unknown-profile diagnostic. The public `full` and `threat-model` benchmark lanes launch with the `exhaustive` profile; the only effective policy change on those lanes is `same_agent_attempts` 3 → 5, since lane and target pins outrank the remaining profile settings (#792). - **[config] [docs]** Renames the unmodified built-in audit profile from `balanced` to `default` and removes the separate default pointer. Configurations that select `balanced` now fail with the standard unknown-profile diagnostic and must select `default` or omit `audit_profile`. Historical run metadata retains its exact recorded id; eval-history and benchmark reporting do not key cohorts by audit-profile id, so otherwise matching runs remain in the same reporting cohort across the migration boundary (#534). @@ -12,7 +13,7 @@ ### Other changes -- **[runtime] [prompts] [docs]** Runtime report presentations (the verified terminal publication and unchecked reports) now restate elapsed time, models, tokens, and estimated spend from `run.json` and the recorded finish time instead of the snapshot the report agent received when its task started, and unchecked reports render the run's goal-search census instead of always stating unknown coverage. The canonical renderer escapes `](` in prose, so it no longer rejects its own output when byte-preserved upstream prose contains link or image syntax, when an issue title contains a non-ASCII letter, or when the public projection redacts a path followed by `(`; the dead audit-context renderer and the prompt's contradictory `## Audit context` instructions are removed. Markdown for reports without `](` in prose is byte-identical, but terminal publications written by earlier versions read back as `not-checked` once the restated summary differs from their stored bytes (usually at least the elapsed time), and an agent report whose prose held a previously accepted `](` form (a `#anchor` or `../` link, or `\](`) no longer re-verifies (#1151). +- **[runtime] [prompts] [docs]** Runtime report presentations (the verified terminal publication and unchecked reports) restate elapsed time, models, tokens, and estimated spend from `run.json` and the recorded finish time instead of the report agent's task-start snapshot, and unchecked reports render the run's goal-search census instead of always stating unknown coverage. The canonical renderer escapes `](` in prose and in warning codes and keeps only its raw-HTML rule for prose, so byte-preserved upstream text with link or image syntax, a non-ASCII issue title, or a legacy-looking label such as a standalone `Critical` no longer makes it throw and fail the final-report gate on every retry. The dead audit-context renderer and the prompt's contradictory `## Audit context` instructions are removed (#1151). - **[runtime]** `resume --refresh-controller` on a dynamically-expanded run now publishes a task manifest that still re-derives from its sealed base, closing a data-integrity regression. Current-controller rendering keeps binding each plan-time prompt to its authenticated retained snapshot (#980), but that execution-only binding now reaches the task-spec literal alone instead of the compiled task manifest, whose `renderedPromptPath` is normalized back to the sealed launch path recorded in `plan.json`. Previously the refreshed controller published the rebound `prompt-snapshots/.md` path into `smithers/tasks.json`, while `verifyDynamicRuntimeMaterialization` re-derived the same document from the sealed `controls/runtime-base-tasks.json` carrying the launch path; the two fingerprinted differently and every gated command (`sync`, `pause`, replay, fork) failed `WORKFLOW_CONTROL_EVIDENCE_INVALID` for the rest of the run's life, with no self-heal on a repeat refresh. Because resume is ungated, a run already damaged this way is recoverable: the normalization scrubs a rebound manifest on the next refresh, including when the sealed base manifest is absent and the live document is the compile input. Prompt authentication is unchanged -- a retained snapshot whose bytes no longer match `plan.json`'s recorded digest still aborts the refresh, and a task with no plan row still fails closed rather than falling back to the cleanup-owned launch path (#1002). - Pi model profiles now accept and forward the CLI's `max` thinking level, in addition to the previously supported levels through `xhigh`. - Upgrades the pinned workflow engine to Smithers 0.35.0. The Effect 4 tree is unchanged at `4.0.0-beta.105`, so the `@effect/*` override family and the `pnpm-workspace.yaml` mirror keep their exact pins; the deterministic npm resolution cutoff moves to `2026-08-18T06:00:00Z`, past `smthrs@0.35.0`'s publish instant. Six compatibility patches are re-derived against upstream's new `smithersRuntimeSpawn`/`smithersRuntimeReentry` indirection and the re-nested `adapter.insertRun` call; none is retired, because every workaround they encode is still absent upstream. Ultrafuzz's Smithers state mirrors gain the new `succeeded-with-failures` run state and `stalled` node state, and the inspect contract admits `tokenUsage`, `run.cancellationSource` and `runState.warnings`. Existing `0.34.0` dependency manifests migrate forward in place at `init`, so an already-scaffolded project converges onto the new pin without `--force` regenerating the run. Native resume never read the target project's `.smithers/package.json` -- launch admission asserts the manifest only against the operator controller's own freshly rendered temporary root -- so a stopped run already resumed under its own run ID and continues to. Ultrafuzz's strict readers for `why`, `status`, `node` and the `TokenUsageReported` event stream also admit the release's new `warnings`, `counts.stalled`, `tokenUsage.freshInputTokens`, `freshInputTokens`/`costUsd` and `stalled` blocker fields, and its two new `NodeStalled`/`RunConcurrencySaturated` event types; a stalled node now resets under `resume --retry-failed` alongside failed ones, and the evmbench adapter no longer aborts a converged run on the release's widened `degraded` verdict. The runner adds four forward-only SQLite migrations (0041-0044) (#955). From 36b70679906ef3abdd23dcb1fd9f14f461da96a4 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 02:51:00 +0000 Subject: [PATCH 8/9] docs(changelog): say the restated summary keeps values run.json lacks The entry said runtime presentations restate the run summary from run.json, but not that a value those records do not carry keeps the agent's copy. docs/reference/artifacts-reports.md already describes that fallback; the entry now says so too. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index e1e34b60b..7f2d61415 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -13,7 +13,7 @@ ### Other changes -- **[runtime] [prompts] [docs]** Runtime report presentations (the verified terminal publication and unchecked reports) restate elapsed time, models, tokens, and estimated spend from `run.json` and the recorded finish time instead of the report agent's task-start snapshot, and unchecked reports render the run's goal-search census instead of always stating unknown coverage. The canonical renderer escapes `](` in prose and in warning codes and keeps only its raw-HTML rule for prose, so byte-preserved upstream text with link or image syntax, a non-ASCII issue title, or a legacy-looking label such as a standalone `Critical` no longer makes it throw and fail the final-report gate on every retry. The dead audit-context renderer and the prompt's contradictory `## Audit context` instructions are removed (#1151). +- **[runtime] [prompts] [docs]** Runtime report presentations (the verified terminal publication and unchecked reports) restate elapsed time, models, tokens, and estimated spend from `run.json` and the recorded finish time, where those records carry them, instead of the report agent's task-start snapshot, and unchecked reports render the run's goal-search census instead of always stating unknown coverage. The canonical renderer escapes `](` in prose and in warning codes and keeps only its raw-HTML rule for prose, so byte-preserved upstream text with link or image syntax, a non-ASCII issue title, or a legacy-looking label such as a standalone `Critical` no longer makes it throw and fail the final-report gate on every retry. The dead audit-context renderer and the prompt's contradictory `## Audit context` instructions are removed (#1151). - **[runtime]** `resume --refresh-controller` on a dynamically-expanded run now publishes a task manifest that still re-derives from its sealed base, closing a data-integrity regression. Current-controller rendering keeps binding each plan-time prompt to its authenticated retained snapshot (#980), but that execution-only binding now reaches the task-spec literal alone instead of the compiled task manifest, whose `renderedPromptPath` is normalized back to the sealed launch path recorded in `plan.json`. Previously the refreshed controller published the rebound `prompt-snapshots/.md` path into `smithers/tasks.json`, while `verifyDynamicRuntimeMaterialization` re-derived the same document from the sealed `controls/runtime-base-tasks.json` carrying the launch path; the two fingerprinted differently and every gated command (`sync`, `pause`, replay, fork) failed `WORKFLOW_CONTROL_EVIDENCE_INVALID` for the rest of the run's life, with no self-heal on a repeat refresh. Because resume is ungated, a run already damaged this way is recoverable: the normalization scrubs a rebound manifest on the next refresh, including when the sealed base manifest is absent and the live document is the compile input. Prompt authentication is unchanged -- a retained snapshot whose bytes no longer match `plan.json`'s recorded digest still aborts the refresh, and a task with no plan row still fails closed rather than falling back to the cleanup-owned launch path (#1002). - Pi model profiles now accept and forward the CLI's `max` thinking level, in addition to the previously supported levels through `xhigh`. - Upgrades the pinned workflow engine to Smithers 0.35.0. The Effect 4 tree is unchanged at `4.0.0-beta.105`, so the `@effect/*` override family and the `pnpm-workspace.yaml` mirror keep their exact pins; the deterministic npm resolution cutoff moves to `2026-08-18T06:00:00Z`, past `smthrs@0.35.0`'s publish instant. Six compatibility patches are re-derived against upstream's new `smithersRuntimeSpawn`/`smithersRuntimeReentry` indirection and the re-nested `adapter.insertRun` call; none is retired, because every workaround they encode is still absent upstream. Ultrafuzz's Smithers state mirrors gain the new `succeeded-with-failures` run state and `stalled` node state, and the inspect contract admits `tokenUsage`, `run.cancellationSource` and `runState.warnings`. Existing `0.34.0` dependency manifests migrate forward in place at `init`, so an already-scaffolded project converges onto the new pin without `--force` regenerating the run. Native resume never read the target project's `.smithers/package.json` -- launch admission asserts the manifest only against the operator controller's own freshly rendered temporary root -- so a stopped run already resumed under its own run ID and continues to. Ultrafuzz's strict readers for `why`, `status`, `node` and the `TokenUsageReported` event stream also admit the release's new `warnings`, `counts.stalled`, `tokenUsage.freshInputTokens`, `freshInputTokens`/`costUsd` and `stalled` blocker fields, and its two new `NodeStalled`/`RunConcurrencySaturated` event types; a stalled node now resets under `resume --retry-failed` alongside failed ones, and the evmbench adapter no longer aborts a converged run on the release's widened `degraded` verdict. The runner adds four forward-only SQLite migrations (0041-0044) (#955). From 4eed4c9538861d98100ebf4b0fa27a110314d719 Mon Sep 17 00:00:00 2001 From: Antonio Viggiano Date: Tue, 29 Sep 2026 03:55:47 +0000 Subject: [PATCH 9/9] chore: move the changelog entry to the consolidated release notes Every pull request in this batch inserts its entry at the same place in CHANGELOG.md, so each merge would conflict with the next. The entries are collected into one changelog update instead. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 2 -- 1 file changed, 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7f2d61415..960ba9394 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,7 +4,6 @@ ### Breaking changes -- **[runtime] [cli] [dashboard] [modal]** Terminal report publications written before this release no longer verify, because readers re-render them and the restated run summary (at least its elapsed time) differs from the stored bytes. `ultrafuzz report` and the dashboard then show a complete run as `# Ultrafuzz report — PARTIAL` with verification `not-checked`, `report bundle` falls back to a report-only archive, and `--require-verified` reads and the Modal public worker fail for that run; nothing re-publishes it automatically. Agent reports whose prose contains `\](` or `](` before an escape-free `../` relative path also render differently and no longer re-verify (#1151). - **[runtime] [cli] [docs]** Restores native Smithers continuation as the normal `resume` behavior. Ordinary resume now sends the persisted workflow and same Smithers run ID directly to Smithers with workflow-change acceptance instead of requiring Ultrafuzz control seals, link journals, controller generations, current schema bindings, graph identity, or metadata projections. `--refresh-controller` renders the current controller beside the historical source and continues that same Smithers run; it no longer publishes an authenticated historical generation. Completed Smithers rows and artifact bytes are not reset, replayed, migrated, or rewritten, and missing optional final-report metadata renders as `unavailable`. Replay determinism after accepting changed workflow source is therefore Smithers' caller-visible responsibility. The controller re-finalization CLI option is removed; use native resume or the existing explicit reset/retry operations (#939). - **[config] [docs] [evals]** Removes the `full` audit profile and renames its packaged topology to `topologies/exhaustive.yml` (byte-identical, topology digest unchanged); the `exhaustive` profile now runs that complete specialist workflow at its existing maximum settings, and no longer follows an edited `.ultrafuzz/topology.yml` (a project `topology_path` or `--topology-path` still overrides it). Configurations that select `full` now fail with the standard unknown-profile diagnostic. The public `full` and `threat-model` benchmark lanes launch with the `exhaustive` profile; the only effective policy change on those lanes is `same_agent_attempts` 3 → 5, since lane and target pins outrank the remaining profile settings (#792). - **[config] [docs]** Renames the unmodified built-in audit profile from `balanced` to `default` and removes the separate default pointer. Configurations that select `balanced` now fail with the standard unknown-profile diagnostic and must select `default` or omit `audit_profile`. Historical run metadata retains its exact recorded id; eval-history and benchmark reporting do not key cohorts by audit-profile id, so otherwise matching runs remain in the same reporting cohort across the migration boundary (#534). @@ -13,7 +12,6 @@ ### Other changes -- **[runtime] [prompts] [docs]** Runtime report presentations (the verified terminal publication and unchecked reports) restate elapsed time, models, tokens, and estimated spend from `run.json` and the recorded finish time, where those records carry them, instead of the report agent's task-start snapshot, and unchecked reports render the run's goal-search census instead of always stating unknown coverage. The canonical renderer escapes `](` in prose and in warning codes and keeps only its raw-HTML rule for prose, so byte-preserved upstream text with link or image syntax, a non-ASCII issue title, or a legacy-looking label such as a standalone `Critical` no longer makes it throw and fail the final-report gate on every retry. The dead audit-context renderer and the prompt's contradictory `## Audit context` instructions are removed (#1151). - **[runtime]** `resume --refresh-controller` on a dynamically-expanded run now publishes a task manifest that still re-derives from its sealed base, closing a data-integrity regression. Current-controller rendering keeps binding each plan-time prompt to its authenticated retained snapshot (#980), but that execution-only binding now reaches the task-spec literal alone instead of the compiled task manifest, whose `renderedPromptPath` is normalized back to the sealed launch path recorded in `plan.json`. Previously the refreshed controller published the rebound `prompt-snapshots/.md` path into `smithers/tasks.json`, while `verifyDynamicRuntimeMaterialization` re-derived the same document from the sealed `controls/runtime-base-tasks.json` carrying the launch path; the two fingerprinted differently and every gated command (`sync`, `pause`, replay, fork) failed `WORKFLOW_CONTROL_EVIDENCE_INVALID` for the rest of the run's life, with no self-heal on a repeat refresh. Because resume is ungated, a run already damaged this way is recoverable: the normalization scrubs a rebound manifest on the next refresh, including when the sealed base manifest is absent and the live document is the compile input. Prompt authentication is unchanged -- a retained snapshot whose bytes no longer match `plan.json`'s recorded digest still aborts the refresh, and a task with no plan row still fails closed rather than falling back to the cleanup-owned launch path (#1002). - Pi model profiles now accept and forward the CLI's `max` thinking level, in addition to the previously supported levels through `xhigh`. - Upgrades the pinned workflow engine to Smithers 0.35.0. The Effect 4 tree is unchanged at `4.0.0-beta.105`, so the `@effect/*` override family and the `pnpm-workspace.yaml` mirror keep their exact pins; the deterministic npm resolution cutoff moves to `2026-08-18T06:00:00Z`, past `smthrs@0.35.0`'s publish instant. Six compatibility patches are re-derived against upstream's new `smithersRuntimeSpawn`/`smithersRuntimeReentry` indirection and the re-nested `adapter.insertRun` call; none is retired, because every workaround they encode is still absent upstream. Ultrafuzz's Smithers state mirrors gain the new `succeeded-with-failures` run state and `stalled` node state, and the inspect contract admits `tokenUsage`, `run.cancellationSource` and `runState.warnings`. Existing `0.34.0` dependency manifests migrate forward in place at `init`, so an already-scaffolded project converges onto the new pin without `--force` regenerating the run. Native resume never read the target project's `.smithers/package.json` -- launch admission asserts the manifest only against the operator controller's own freshly rendered temporary root -- so a stopped run already resumed under its own run ID and continues to. Ultrafuzz's strict readers for `why`, `status`, `node` and the `TokenUsageReported` event stream also admit the release's new `warnings`, `counts.stalled`, `tokenUsage.freshInputTokens`, `freshInputTokens`/`costUsd` and `stalled` blocker fields, and its two new `NodeStalled`/`RunConcurrencySaturated` event types; a stalled node now resets under `resume --retry-failed` alongside failed ones, and the evmbench adapter no longer aborts a converged run on the release's widened `degraded` verdict. The runner adds four forward-only SQLite migrations (0041-0044) (#955).